{"schemaVersion":"1.0.0","dataset":"benchmarks","updatedAt":"2026-08-24","recordCount":12653,"records":[{"resultId":"evidence-2026-07-001","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":55.1,"normalizedScore":55.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-002","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":83.6,"normalizedScore":83.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-003","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":56.5,"normalizedScore":56.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-004","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":78.4,"normalizedScore":78.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-005","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"v2","score":57.9,"normalizedScore":57.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-006","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":84.2,"normalizedScore":84.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-007","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":83.6,"normalizedScore":83.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-008","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":40.2,"normalizedScore":40.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-009","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2","score":72.1,"normalizedScore":72.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-010","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"blueprint-bench-2","benchmarkName":"Blueprint-Bench 2","benchmarkCategory":"multimodal","benchmarkOrganisation":"Google","benchmarkVersion":"v2","score":33.6,"normalizedScore":33.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-011","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":54.2,"normalizedScore":54.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-012","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":78.2,"normalizedScore":78.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-013","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":76.2,"normalizedScore":76.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-014","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"v2","score":43,"normalizedScore":43,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-015","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":83.3,"normalizedScore":83.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-016","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":80.5,"normalizedScore":80.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-017","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":44.4,"normalizedScore":44.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-018","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2","score":77.1,"normalizedScore":77.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-019","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"blueprint-bench-2","benchmarkName":"Blueprint-Bench 2","benchmarkCategory":"multimodal","benchmarkOrganisation":"Google","benchmarkVersion":"v2","score":26.5,"normalizedScore":26.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-020","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":58.6,"normalizedScore":58.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-021","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":75.3,"normalizedScore":75.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-022","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":55.6,"normalizedScore":55.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-023","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":78.7,"normalizedScore":78.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-024","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"v2","score":51.8,"normalizedScore":51.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-025","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":84.1,"normalizedScore":84.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-026","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":81.2,"normalizedScore":81.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-027","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":41.4,"normalizedScore":41.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-028","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2","score":84.6,"normalizedScore":84.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-029","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"blueprint-bench-2","benchmarkName":"Blueprint-Bench 2","benchmarkCategory":"multimodal","benchmarkOrganisation":"Google","benchmarkVersion":"v2","score":36.2,"normalizedScore":36.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant and provider evaluation configuration stated in the Gemini 3.5 Flash model card; single-attempt where specified.","methodologyVersion":"1.0.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-030","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":76.2,"normalizedScore":76.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-031","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":70.3,"normalizedScore":70.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-032","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":78.2,"normalizedScore":78.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Terminus-2 agent system; provider-published Terminal-Bench 2.1 evaluation.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-19","publishedAt":"2026-05-19","sourceId":"google-gemini-35-model-card","sourceTitle":"Gemini 3.5 Flash model card","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/model-cards/gemini-3-5-flash/","sourceDate":"2026-05-19","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-033","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":83.82,"normalizedScore":83.82,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Code; reasoning xhigh; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-06-07","publishedAt":"2026-06-07","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-06-07","sourceCheckedAt":"2026-07-16","checkedAt":"2026-08-16","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-034","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":80.45,"normalizedScore":80.45,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-06-05","publishedAt":"2026-06-05","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-06-05","sourceCheckedAt":"2026-07-16","checkedAt":"2026-08-16","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-035","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":83.15,"normalizedScore":83.15,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Codex; reasoning xhigh; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-01","publishedAt":"2026-05-01","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-05-01","sourceCheckedAt":"2026-07-16","checkedAt":"2026-08-16","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-036","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":77.98,"normalizedScore":77.98,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Terminus 2; reasoning xhigh; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-01","publishedAt":"2026-05-01","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-05-01","sourceCheckedAt":"2026-07-16","checkedAt":"2026-08-16","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-037","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":79.33,"normalizedScore":79.33,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Cursor CLI; reasoning high; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-08-16","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-038","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":78.88,"normalizedScore":78.88,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Code; reasoning high; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-08-16","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-039","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":78.43,"normalizedScore":78.43,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Codex; reasoning max; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-11","publishedAt":"2026-07-11","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-07-11","sourceCheckedAt":"2026-07-16","checkedAt":"2026-08-16","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-040","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":75.73,"normalizedScore":75.73,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Codex; reasoning max; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-11","publishedAt":"2026-07-11","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-07-11","sourceCheckedAt":"2026-07-16","checkedAt":"2026-08-16","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-041","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":74.61,"normalizedScore":74.61,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Code; reasoning high; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-08-16","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-042","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":65.8,"normalizedScore":65.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini CLI; reasoning high; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-01","publishedAt":"2026-05-01","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-05-01","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-043","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":65.6,"normalizedScore":65.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.0.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-01","publishedAt":"2026-05-01","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-05-01","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-044","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":47.17,"normalizedScore":47.17,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-sol-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-sol","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-045","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":94.14,"normalizedScore":94.14,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-sol-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-sol","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-046","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":83.41,"normalizedScore":83.41,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-sol-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-sol","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-047","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":88.01,"normalizedScore":88.01,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-sol-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-sol","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-sol","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-048","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":41.8,"normalizedScore":41.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-terra-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-terra","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-049","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.53,"normalizedScore":92.53,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-terra-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-terra","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-050","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":80.69,"normalizedScore":80.69,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-terra-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-terra","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-051","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":88.01,"normalizedScore":88.01,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-terra-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-terra","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-terra","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-052","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":37.21,"normalizedScore":37.21,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-luna-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-luna","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-053","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":91.11,"normalizedScore":91.11,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-luna-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-luna","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-054","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":78.55,"normalizedScore":78.55,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-luna-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-luna","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-055","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":80.9,"normalizedScore":80.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-6-luna-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-6-luna","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-6-luna","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-056","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":53.34,"normalizedScore":53.34,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-fable-5-evals","sourceTitle":"Artificial Analysis evaluations for claude-fable-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-057","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.63,"normalizedScore":92.63,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-fable-5-evals","sourceTitle":"Artificial Analysis evaluations for claude-fable-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-058","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":84.64,"normalizedScore":84.64,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-fable-5-evals","sourceTitle":"Artificial Analysis evaluations for claude-fable-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-fable-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-059","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":45.74,"normalizedScore":45.74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-opus-4-8-evals","sourceTitle":"Artificial Analysis evaluations for claude-opus-4-8","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-4-8","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-060","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.02,"normalizedScore":92.02,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-opus-4-8-evals","sourceTitle":"Artificial Analysis evaluations for claude-opus-4-8","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-4-8","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-061","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":84.64,"normalizedScore":84.64,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-opus-4-8-evals","sourceTitle":"Artificial Analysis evaluations for claude-opus-4-8","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-opus-4-8","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-062","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":39.57,"normalizedScore":39.57,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-sonnet-5-evals","sourceTitle":"Artificial Analysis evaluations for claude-sonnet-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-sonnet-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-063","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":91.11,"normalizedScore":91.11,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-sonnet-5-evals","sourceTitle":"Artificial Analysis evaluations for claude-sonnet-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-sonnet-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-064","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":77.28,"normalizedScore":77.28,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-sonnet-5-evals","sourceTitle":"Artificial Analysis evaluations for claude-sonnet-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-sonnet-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-065","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":80.52,"normalizedScore":80.52,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-sonnet-5-evals","sourceTitle":"Artificial Analysis evaluations for claude-sonnet-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-sonnet-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-066","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":4.26,"normalizedScore":4.26,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-4-5-haiku-evals","sourceTitle":"Artificial Analysis evaluations for claude-4-5-haiku","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-4-5-haiku","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-067","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":64.65,"normalizedScore":64.65,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-4-5-haiku-evals","sourceTitle":"Artificial Analysis evaluations for claude-4-5-haiku","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-4-5-haiku","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-068","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":55.14,"normalizedScore":55.14,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-4-5-haiku-evals","sourceTitle":"Artificial Analysis evaluations for claude-4-5-haiku","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-4-5-haiku","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-069","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":51.11,"normalizedScore":51.11,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-claude-4-5-haiku-evals","sourceTitle":"Artificial Analysis evaluations for claude-4-5-haiku","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/claude-4-5-haiku","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-070","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":44.72,"normalizedScore":44.72,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gemini-3-1-pro-preview-evals","sourceTitle":"Artificial Analysis evaluations for gemini-3-1-pro-preview","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-1-pro-preview","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-071","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":94.14,"normalizedScore":94.14,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gemini-3-1-pro-preview-evals","sourceTitle":"Artificial Analysis evaluations for gemini-3-1-pro-preview","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-1-pro-preview","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-072","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":82.43,"normalizedScore":82.43,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gemini-3-1-pro-preview-evals","sourceTitle":"Artificial Analysis evaluations for gemini-3-1-pro-preview","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-1-pro-preview","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-073","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":73.78,"normalizedScore":73.78,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gemini-3-1-pro-preview-evals","sourceTitle":"Artificial Analysis evaluations for gemini-3-1-pro-preview","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-1-pro-preview","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-074","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":35.03,"normalizedScore":35.03,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-grok-4-3-evals","sourceTitle":"Artificial Analysis evaluations for grok-4-3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/grok-4-3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-075","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":90.1,"normalizedScore":90.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-grok-4-3-evals","sourceTitle":"Artificial Analysis evaluations for grok-4-3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/grok-4-3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-076","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":78.09,"normalizedScore":78.09,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-grok-4-3-evals","sourceTitle":"Artificial Analysis evaluations for grok-4-3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/grok-4-3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-077","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":39.7,"normalizedScore":39.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-grok-4-3-evals","sourceTitle":"Artificial Analysis evaluations for grok-4-3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/grok-4-3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-078","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":35.87,"normalizedScore":35.87,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-deepseek-v4-pro-evals","sourceTitle":"Artificial Analysis evaluations for deepseek-v4-pro","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-079","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":88.79,"normalizedScore":88.79,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-deepseek-v4-pro-evals","sourceTitle":"Artificial Analysis evaluations for deepseek-v4-pro","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-080","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":64.04,"normalizedScore":64.04,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-deepseek-v4-pro-evals","sourceTitle":"Artificial Analysis evaluations for deepseek-v4-pro","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-pro","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-081","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":32.07,"normalizedScore":32.07,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-deepseek-v4-flash-evals","sourceTitle":"Artificial Analysis evaluations for deepseek-v4-flash","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-flash","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-082","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":89.39,"normalizedScore":89.39,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-deepseek-v4-flash-evals","sourceTitle":"Artificial Analysis evaluations for deepseek-v4-flash","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-flash","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-083","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":61.8,"normalizedScore":61.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-deepseek-v4-flash-evals","sourceTitle":"Artificial Analysis evaluations for deepseek-v4-flash","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/deepseek-v4-flash","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-084","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":12.79,"normalizedScore":12.79,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-medium-3-5-evals","sourceTitle":"Artificial Analysis evaluations for mistral-medium-3-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-085","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":74.85,"normalizedScore":74.85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-medium-3-5-evals","sourceTitle":"Artificial Analysis evaluations for mistral-medium-3-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-086","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":64.86,"normalizedScore":64.86,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-medium-3-5-evals","sourceTitle":"Artificial Analysis evaluations for mistral-medium-3-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-087","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":50.56,"normalizedScore":50.56,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-medium-3-5-evals","sourceTitle":"Artificial Analysis evaluations for mistral-medium-3-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-medium-3-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-088","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":9.5,"normalizedScore":9.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-small-4-evals","sourceTitle":"Artificial Analysis evaluations for mistral-small-4","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-small-4","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-089","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":76.87,"normalizedScore":76.87,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-small-4-evals","sourceTitle":"Artificial Analysis evaluations for mistral-small-4","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-small-4","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-090","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":56.82,"normalizedScore":56.82,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-small-4-evals","sourceTitle":"Artificial Analysis evaluations for mistral-small-4","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-small-4","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-091","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":20.97,"normalizedScore":20.97,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-small-4-evals","sourceTitle":"Artificial Analysis evaluations for mistral-small-4","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-small-4","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-092","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":4.08,"normalizedScore":4.08,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-large-3-evals","sourceTitle":"Artificial Analysis evaluations for mistral-large-3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-large-3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-093","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":67.98,"normalizedScore":67.98,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-large-3-evals","sourceTitle":"Artificial Analysis evaluations for mistral-large-3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-large-3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-094","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":55.66,"normalizedScore":55.66,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-large-3-evals","sourceTitle":"Artificial Analysis evaluations for mistral-large-3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-large-3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-095","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":46.46,"normalizedScore":46.46,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-large-3-evals","sourceTitle":"Artificial Analysis evaluations for mistral-large-3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-large-3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-096","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":11.99,"normalizedScore":11.99,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-mistral-large-3-evals","sourceTitle":"Artificial Analysis evaluations for mistral-large-3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/mistral-large-3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-097","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":38.09,"normalizedScore":38.09,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-qwen3-7-max-evals","sourceTitle":"Artificial Analysis evaluations for qwen3-7-max","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/qwen3-7-max","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-098","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.32,"normalizedScore":92.32,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-qwen3-7-max-evals","sourceTitle":"Artificial Analysis evaluations for qwen3-7-max","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/qwen3-7-max","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-099","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":74.53,"normalizedScore":74.53,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-qwen3-7-max-evals","sourceTitle":"Artificial Analysis evaluations for qwen3-7-max","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/qwen3-7-max","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-100","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":41.61,"normalizedScore":41.61,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-4-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-4","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-4","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-101","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.02,"normalizedScore":92.02,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-4-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-4","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-4","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-102","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":78.44,"normalizedScore":78.44,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-4-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-4","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-4","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-103","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":78.28,"normalizedScore":78.28,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-4-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-4","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-4","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-104","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":39.9,"normalizedScore":39.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-3-codex-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-3-codex","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-3-codex","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-105","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":91.52,"normalizedScore":91.52,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-3-codex-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-3-codex","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-3-codex","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-106","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":78.5,"normalizedScore":78.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-3-codex-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-3-codex","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-3-codex","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-107","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":37.12,"normalizedScore":37.12,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-minimax-m3-evals","sourceTitle":"Artificial Analysis evaluations for minimax-m3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/minimax-m3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-108","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.93,"normalizedScore":92.93,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-minimax-m3-evals","sourceTitle":"Artificial Analysis evaluations for minimax-m3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/minimax-m3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-109","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":78.55,"normalizedScore":78.55,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-minimax-m3-evals","sourceTitle":"Artificial Analysis evaluations for minimax-m3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/minimax-m3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-110","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":65.17,"normalizedScore":65.17,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-minimax-m3-evals","sourceTitle":"Artificial Analysis evaluations for minimax-m3","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/minimax-m3","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-111","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":44.3,"normalizedScore":44.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-5-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-112","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":93.54,"normalizedScore":93.54,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-5-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-113","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":79.88,"normalizedScore":79.88,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-5-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-114","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":84.27,"normalizedScore":84.27,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gpt-5-5-evals","sourceTitle":"Artificial Analysis evaluations for gpt-5-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gpt-5-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-115","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":39.85,"normalizedScore":39.85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gemini-3-5-flash-medium-evals","sourceTitle":"Artificial Analysis evaluations for gemini-3-5-flash-medium","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-medium","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-116","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.12,"normalizedScore":92.12,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gemini-3-5-flash-medium-evals","sourceTitle":"Artificial Analysis evaluations for gemini-3-5-flash-medium","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-medium","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-117","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":83.87,"normalizedScore":83.87,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gemini-3-5-flash-medium-evals","sourceTitle":"Artificial Analysis evaluations for gemini-3-5-flash-medium","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-medium","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-118","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":40.27,"normalizedScore":40.27,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-grok-4-5-evals","sourceTitle":"Artificial Analysis evaluations for grok-4-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/grok-4-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-119","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":93.13,"normalizedScore":93.13,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-grok-4-5-evals","sourceTitle":"Artificial Analysis evaluations for grok-4-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/grok-4-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-120","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":80.4,"normalizedScore":80.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-grok-4-5-evals","sourceTitle":"Artificial Analysis evaluations for grok-4-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/grok-4-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-121","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":81.65,"normalizedScore":81.65,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis public model evaluation page; exact provider variant as listed on the page.","methodologyVersion":"1.1.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-grok-4-5-evals","sourceTitle":"Artificial Analysis evaluations for grok-4-5","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/grok-4-5","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-122","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":64.5,"normalizedScore":64.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-123","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":57.9,"normalizedScore":57.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-124","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":57.4,"normalizedScore":57.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-125","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":52.2,"normalizedScore":52.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-126","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":52.1,"normalizedScore":52.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-127","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":41.4,"normalizedScore":41.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-128","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":40.2,"normalizedScore":40.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-129","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":35,"normalizedScore":35,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-130","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":34.7,"normalizedScore":34.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-131","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":30.1,"normalizedScore":30.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-132","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":8.1,"normalizedScore":8.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-133","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":7.7,"normalizedScore":7.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-134","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":94.6,"normalizedScore":94.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-135","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":94.1,"normalizedScore":94.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-136","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":93.6,"normalizedScore":93.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-137","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":93.6,"normalizedScore":93.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-138","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.9,"normalizedScore":92.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-139","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.8,"normalizedScore":92.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-140","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.4,"normalizedScore":92.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-141","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.3,"normalizedScore":92.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-142","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.2,"normalizedScore":92.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-143","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":90.3,"normalizedScore":90.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-144","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":90.1,"normalizedScore":90.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-145","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":87.6,"normalizedScore":87.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-146","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":72.9,"normalizedScore":72.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-147","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":71.2,"normalizedScore":71.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-148","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":91.9,"normalizedScore":91.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-149","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":88,"normalizedScore":88,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-150","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":87.4,"normalizedScore":87.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-151","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":84.7,"normalizedScore":84.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-152","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":84.3,"normalizedScore":84.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-153","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":83.3,"normalizedScore":83.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-154","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":82,"normalizedScore":82,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-155","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":80.4,"normalizedScore":80.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-156","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":76.2,"normalizedScore":76.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-157","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":74.6,"normalizedScore":74.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-158","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":70.3,"normalizedScore":70.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-159","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":69.7,"normalizedScore":69.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-160","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":66,"normalizedScore":66,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-161","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":59.1,"normalizedScore":59.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-162","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":49.1,"normalizedScore":49.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-163","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":80.3,"normalizedScore":80.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-164","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":80,"normalizedScore":80,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-165","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":69.2,"normalizedScore":69.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-166","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":64.7,"normalizedScore":64.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-167","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":64.6,"normalizedScore":64.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-168","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":63.4,"normalizedScore":63.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-169","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":63.2,"normalizedScore":63.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-170","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":62.7,"normalizedScore":62.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-171","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":60.6,"normalizedScore":60.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-172","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":59,"normalizedScore":59,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-173","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":58.6,"normalizedScore":58.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-174","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":57.7,"normalizedScore":57.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-175","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":57.6,"normalizedScore":57.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-176","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":56.8,"normalizedScore":56.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-177","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":55.1,"normalizedScore":55.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-178","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":52.1,"normalizedScore":52.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-179","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":50.7,"normalizedScore":50.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-180","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":49.1,"normalizedScore":49.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-181","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":95.5,"normalizedScore":95.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-182","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":95,"normalizedScore":95,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-183","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":88.6,"normalizedScore":88.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-184","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":85.2,"normalizedScore":85.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-185","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":85,"normalizedScore":85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-186","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":80.5,"normalizedScore":80.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-187","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":80.4,"normalizedScore":80.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-188","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":77.7,"normalizedScore":77.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-189","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":76.8,"normalizedScore":76.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-190","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":73.7,"normalizedScore":73.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-191","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":73.6,"normalizedScore":73.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-192","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":73.3,"normalizedScore":73.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-193","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":85,"normalizedScore":85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-194","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":85,"normalizedScore":85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-195","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":83.4,"normalizedScore":83.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-196","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":81.2,"normalizedScore":81.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-197","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":78.7,"normalizedScore":78.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-198","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":78.4,"normalizedScore":78.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-199","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":75,"normalizedScore":75,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-200","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":73.3,"normalizedScore":73.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-201","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":70.06,"normalizedScore":70.06,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-202","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":64.7,"normalizedScore":64.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-203","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":92.2,"normalizedScore":92.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-204","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":88,"normalizedScore":88,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-205","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":87.5,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-206","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":84.7,"normalizedScore":84.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-207","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":84.4,"normalizedScore":84.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-208","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":84.3,"normalizedScore":84.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-209","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":83.52,"normalizedScore":83.52,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-210","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":83.3,"normalizedScore":83.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-211","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":82.7,"normalizedScore":82.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-212","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":60.6,"normalizedScore":60.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-213","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":83.9,"normalizedScore":83.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-214","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":83.6,"normalizedScore":83.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-215","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":83,"normalizedScore":83,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-216","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":81.2,"normalizedScore":81.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-217","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":81.2,"normalizedScore":81.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-218","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":80.7,"normalizedScore":80.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-219","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":79,"normalizedScore":79,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-220","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":78.5,"normalizedScore":78.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-221","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":78.4,"normalizedScore":78.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-222","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":78.1,"normalizedScore":78.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-223","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":78.1,"normalizedScore":78.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-224","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":91.6,"normalizedScore":91.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-225","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":89.6,"normalizedScore":89.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-226","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":85,"normalizedScore":85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-227","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":56.8,"normalizedScore":56.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-228","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":55.2,"normalizedScore":55.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-229","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2","score":85,"normalizedScore":85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-230","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2","score":77.1,"normalizedScore":77.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-231","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2","score":72.1,"normalizedScore":72.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-232","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":93.5,"normalizedScore":93.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-233","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":89.9,"normalizedScore":89.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-234","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":88.3,"normalizedScore":88.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-235","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":85.9,"normalizedScore":85.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-236","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":84.2,"normalizedScore":84.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-237","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":82.8,"normalizedScore":82.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-238","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":80.2,"normalizedScore":80.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-239","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":73.2,"normalizedScore":73.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-240","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":83.6,"normalizedScore":83.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-241","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":82.2,"normalizedScore":82.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-242","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":76.4,"normalizedScore":76.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-243","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":75.3,"normalizedScore":75.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-244","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":74.2,"normalizedScore":74.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-245","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":73.2,"normalizedScore":73.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-246","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":70.6,"normalizedScore":70.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-247","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":69.4,"normalizedScore":69.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-248","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":64,"normalizedScore":64,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-249","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":29.5,"normalizedScore":29.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-250","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":59.9,"normalizedScore":59.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-251","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":58,"normalizedScore":58,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-252","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":56.5,"normalizedScore":56.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-253","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":55.6,"normalizedScore":55.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-254","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":54.6,"normalizedScore":54.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-255","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":53.4,"normalizedScore":53.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-256","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":53.1,"normalizedScore":53.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-257","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":46.3,"normalizedScore":46.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-258","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":40.7,"normalizedScore":40.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-259","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":27.8,"normalizedScore":27.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-260","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":89,"normalizedScore":89,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-261","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":84.9,"normalizedScore":84.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-262","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":78.6,"normalizedScore":78.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-263","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":51.7,"normalizedScore":51.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-264","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":53.5,"normalizedScore":53.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-265","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":53.1,"normalizedScore":53.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-266","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":51.3,"normalizedScore":51.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-267","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":48.7,"normalizedScore":48.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-268","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":47.3,"normalizedScore":47.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-269","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"simpleqa","benchmarkName":"SimpleQA","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":45,"normalizedScore":45,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-270","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"simpleqa","benchmarkName":"SimpleQA","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":23.1,"normalizedScore":23.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-271","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":89.6,"normalizedScore":89.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-272","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":88.5,"normalizedScore":88.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-273","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":87.1,"normalizedScore":87.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-274","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":83,"normalizedScore":83,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-275","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":82.9,"normalizedScore":82.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-276","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":98.5,"normalizedScore":98.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-277","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":97.7,"normalizedScore":97.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-278","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":95.9,"normalizedScore":95.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-279","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":95.6,"normalizedScore":95.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-280","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":95.3,"normalizedScore":95.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-281","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":94.7,"normalizedScore":94.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-282","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":94.4,"normalizedScore":94.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-283","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":93.9,"normalizedScore":93.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-284","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":93,"normalizedScore":93,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-285","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":88.9,"normalizedScore":88.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-286","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":87.1,"normalizedScore":87.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-287","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":86.3,"normalizedScore":86.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-288","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":86,"normalizedScore":86,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-289","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":85.1,"normalizedScore":85.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-290","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":41.2,"normalizedScore":41.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-291","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":31.3,"normalizedScore":31.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-292","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":24.6,"normalizedScore":24.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-293","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":94.6,"normalizedScore":94.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-294","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":94.3,"normalizedScore":94.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-295","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":93.9,"normalizedScore":93.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-296","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2","score":61,"normalizedScore":61,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-297","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":80.4,"normalizedScore":80.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-298","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":78.5,"normalizedScore":78.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-299","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":77.3,"normalizedScore":77.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-300","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":75.5,"normalizedScore":75.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-301","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":56.8,"normalizedScore":56.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-302","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":55.7,"normalizedScore":55.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-303","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"3","score":7.8,"normalizedScore":7.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-304","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"3","score":0.8,"normalizedScore":0.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-305","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"3","score":0.2,"normalizedScore":0.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-306","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":47.6,"normalizedScore":47.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-307","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":47.241,"normalizedScore":47.241,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-308","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":38.966,"normalizedScore":38.966,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-309","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":36.9,"normalizedScore":36.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-310","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":27.9,"normalizedScore":27.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-311","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":5.903,"normalizedScore":5.903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-312","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":60.2,"normalizedScore":60.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-313","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":58.9,"normalizedScore":58.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-314","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":56.6,"normalizedScore":56.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-315","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":56.1,"normalizedScore":56.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-316","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":56.1,"normalizedScore":56.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-317","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":54.1,"normalizedScore":54.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-318","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":53.9,"normalizedScore":53.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-319","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":53.6,"normalizedScore":53.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-320","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":53.5,"normalizedScore":53.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-321","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":53.2,"normalizedScore":53.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-322","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":52.5,"normalizedScore":52.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-323","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":45.4,"normalizedScore":45.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-324","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":41.9,"normalizedScore":41.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-325","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":38,"normalizedScore":38,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-326","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":36.2,"normalizedScore":36.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-327","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":81.3,"normalizedScore":81.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-328","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":76.3,"normalizedScore":76.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-329","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":82.9,"normalizedScore":82.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-330","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":77.2,"normalizedScore":77.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-331","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":77.1,"normalizedScore":77.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-332","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":75.9,"normalizedScore":75.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-333","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":75.4,"normalizedScore":75.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-334","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":73.9,"normalizedScore":73.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-335","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":72.7,"normalizedScore":72.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-336","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":71.2,"normalizedScore":71.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-337","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":63.5,"normalizedScore":63.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-338","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":62.2,"normalizedScore":62.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-339","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":48.2,"normalizedScore":48.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-340","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":36.2,"normalizedScore":36.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-341","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2","score":91.7,"normalizedScore":91.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-342","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2","score":90.4,"normalizedScore":90.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-343","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2","score":77.3,"normalizedScore":77.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-344","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":null,"score":62.6,"normalizedScore":62.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-345","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":null,"score":50.2,"normalizedScore":50.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-346","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":null,"score":45.6,"normalizedScore":45.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-347","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":null,"score":20.6,"normalizedScore":20.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-348","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":null,"score":13,"normalizedScore":13,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-349","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":null,"score":4.6,"normalizedScore":4.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-350","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":null,"score":2.8,"normalizedScore":2.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-351","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026-05","score":33.7,"normalizedScore":33.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-352","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026-05","score":23.2,"normalizedScore":23.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-353","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026-05","score":17.5,"normalizedScore":17.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-354","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026-05","score":13.4,"normalizedScore":13.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-355","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026-05","score":12.4,"normalizedScore":12.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-356","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026-05","score":6,"normalizedScore":6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-357","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"berkeley-function-calling-leaderboard","benchmarkName":"Berkeley Function-Calling Leaderboard","benchmarkCategory":"agents","benchmarkOrganisation":"Berkeley Gorilla","benchmarkVersion":"V3","score":75,"normalizedScore":75,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-358","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"berkeley-function-calling-leaderboard","benchmarkName":"Berkeley Function-Calling Leaderboard","benchmarkCategory":"agents","benchmarkOrganisation":"Berkeley Gorilla","benchmarkVersion":"V3","score":72.9,"normalizedScore":72.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-359","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"blueprint-bench-2","benchmarkName":"Blueprint-Bench 2","benchmarkCategory":"multimodal","benchmarkOrganisation":"Google","benchmarkVersion":"v2","score":38.6,"normalizedScore":38.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-360","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"blueprint-bench-2","benchmarkName":"Blueprint-Bench 2","benchmarkCategory":"multimodal","benchmarkOrganisation":"Google","benchmarkVersion":"v2","score":33.6,"normalizedScore":33.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.2.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-361","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":65.9,"normalizedScore":65.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-362","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":62.9,"normalizedScore":62.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-363","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":60.6,"normalizedScore":60.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-364","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":58.3,"normalizedScore":58.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-365","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":57.6,"normalizedScore":57.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-366","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":57.6,"normalizedScore":57.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-367","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":53.8,"normalizedScore":53.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-368","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":53,"normalizedScore":53,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-369","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":50.8,"normalizedScore":50.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-370","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":47,"normalizedScore":47,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-371","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":42.4,"normalizedScore":42.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-372","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":40.9,"normalizedScore":40.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-373","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":37.9,"normalizedScore":37.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-374","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":34.8,"normalizedScore":34.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-375","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":24.2,"normalizedScore":24.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-376","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":17.4,"normalizedScore":17.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-377","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":15.9,"normalizedScore":15.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on BenchLM (Terminal-Bench Hard mapped to long-horizon terminal family).","methodologyVersion":"1.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-378","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":70.5,"normalizedScore":70.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cursor-bench-public","sourceTitle":"CursorBench public evaluation via BenchLM","sourcePublisher":"Cursor / BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/cursorBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-379","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":67.2,"normalizedScore":67.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cursor-bench-public","sourceTitle":"CursorBench public evaluation via BenchLM","sourcePublisher":"Cursor / BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/cursorBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-380","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":66.7,"normalizedScore":66.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cursor-bench-public","sourceTitle":"CursorBench public evaluation via BenchLM","sourcePublisher":"Cursor / BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/cursorBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-381","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":64.9,"normalizedScore":64.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cursor-bench-public","sourceTitle":"CursorBench public evaluation via BenchLM","sourcePublisher":"Cursor / BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/cursorBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-382","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":62.3,"normalizedScore":62.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cursor-bench-public","sourceTitle":"CursorBench public evaluation via BenchLM","sourcePublisher":"Cursor / BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/cursorBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-383","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":61.5,"normalizedScore":61.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cursor-bench-public","sourceTitle":"CursorBench public evaluation via BenchLM","sourcePublisher":"Cursor / BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/cursorBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-384","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":61.1,"normalizedScore":61.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cursor-bench-public","sourceTitle":"CursorBench public evaluation via BenchLM","sourcePublisher":"Cursor / BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/cursorBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-385","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":58.4,"normalizedScore":58.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cursor-bench-public","sourceTitle":"CursorBench public evaluation via BenchLM","sourcePublisher":"Cursor / BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/cursorBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-386","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":55,"normalizedScore":55,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cursor-bench-public","sourceTitle":"CursorBench public evaluation via BenchLM","sourcePublisher":"Cursor / BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/cursorBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-387","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":48.8,"normalizedScore":48.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cursor-bench-public","sourceTitle":"CursorBench public evaluation via BenchLM","sourcePublisher":"Cursor / BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/cursorBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-388","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":77.09,"normalizedScore":77.09,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-389","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":81.95,"normalizedScore":81.95,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-390","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":87.8,"normalizedScore":87.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-391","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":92.8,"normalizedScore":92.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-392","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":82,"normalizedScore":82,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-393","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":76.84,"normalizedScore":76.84,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-394","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":81.7,"normalizedScore":81.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-395","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":89,"normalizedScore":89,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-396","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":79.7,"normalizedScore":79.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-397","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":83.1,"normalizedScore":83.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-398","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":75.49,"normalizedScore":75.49,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-399","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":74.08,"normalizedScore":74.08,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-400","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":77,"normalizedScore":77,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-401","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":75.4,"normalizedScore":75.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-402","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":95.3,"normalizedScore":95.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-403","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":68.06,"normalizedScore":68.06,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-404","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":73.49,"normalizedScore":73.49,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-405","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":88.1,"normalizedScore":88.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-406","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":66.2,"normalizedScore":66.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-407","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":73.6,"normalizedScore":73.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-408","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":70.87,"normalizedScore":70.87,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-409","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":69.1,"normalizedScore":69.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-410","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":88.3,"normalizedScore":88.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-411","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":75.3,"normalizedScore":75.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-412","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":62.09,"normalizedScore":62.09,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-413","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":71.25,"normalizedScore":71.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-414","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":69.7,"normalizedScore":69.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-415","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":71.38,"normalizedScore":71.38,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-416","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":81.9,"normalizedScore":81.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-417","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":82.2,"normalizedScore":82.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-418","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":58.3,"normalizedScore":58.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-419","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":70.3,"normalizedScore":70.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-420","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":49.1,"normalizedScore":49.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-421","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":65.4,"normalizedScore":65.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-422","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":72.7,"normalizedScore":72.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-423","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":71.9,"normalizedScore":71.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-424","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":75.6,"normalizedScore":75.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-425","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":61.7,"normalizedScore":61.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-426","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":59.82,"normalizedScore":59.82,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-427","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":64.17,"normalizedScore":64.17,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-428","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":87.5,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-429","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":96.8,"normalizedScore":96.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-430","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":60,"normalizedScore":60,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-431","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":70.2,"normalizedScore":70.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-432","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":67.83,"normalizedScore":67.83,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-433","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":66.59,"normalizedScore":66.59,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-434","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":75.6,"normalizedScore":75.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-435","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":72.5,"normalizedScore":72.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-436","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":95.3,"normalizedScore":95.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-437","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":66.06,"normalizedScore":66.06,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-438","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":69.36,"normalizedScore":69.36,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-439","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":65.7,"normalizedScore":65.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-440","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":59.81,"normalizedScore":59.81,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-441","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":66.36,"normalizedScore":66.36,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-442","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":50.4,"normalizedScore":50.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-443","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":79.2,"normalizedScore":79.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-444","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":74.6,"normalizedScore":74.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-445","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":55.5,"normalizedScore":55.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-446","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":43.12,"normalizedScore":43.12,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-447","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":55.16,"normalizedScore":55.16,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-448","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":77.4,"normalizedScore":77.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-449","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":49.75,"normalizedScore":49.75,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-450","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":58.06,"normalizedScore":58.06,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-451","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":49.6,"normalizedScore":49.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-452","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":68.8,"normalizedScore":68.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-453","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":58.87,"normalizedScore":58.87,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-454","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":65.66,"normalizedScore":65.66,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-455","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":87.8,"normalizedScore":87.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-456","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":91.5,"normalizedScore":91.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-457","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":75.4,"normalizedScore":75.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-458","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":54,"normalizedScore":54,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-459","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":49.36,"normalizedScore":49.36,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-460","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":56.67,"normalizedScore":56.67,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-461","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":56.2,"normalizedScore":56.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-462","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":83.7,"normalizedScore":83.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-463","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":64.4,"normalizedScore":64.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-464","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":63.63,"normalizedScore":63.63,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-465","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":71,"normalizedScore":71,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-466","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":59.9,"normalizedScore":59.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-467","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":63.85,"normalizedScore":63.85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-468","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":81.8,"normalizedScore":81.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-469","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":77.1,"normalizedScore":77.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-470","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":80.3,"normalizedScore":80.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-471","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":59.7,"normalizedScore":59.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-472","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":64.62,"normalizedScore":64.62,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-473","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":64.26,"normalizedScore":64.26,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-474","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":75.2,"normalizedScore":75.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-475","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":69.5,"normalizedScore":69.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-476","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":95.3,"normalizedScore":95.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-477","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":54.2,"normalizedScore":54.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-478","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":51.6,"normalizedScore":51.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-479","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":29.3,"normalizedScore":29.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-480","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":67.6,"normalizedScore":67.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-481","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":53.9,"normalizedScore":53.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-482","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":49,"normalizedScore":49,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-483","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":61.78,"normalizedScore":61.78,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-484","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":64.08,"normalizedScore":64.08,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-485","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":87.4,"normalizedScore":87.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-486","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":89,"normalizedScore":89,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-487","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":91.4,"normalizedScore":91.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-488","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":54.85,"normalizedScore":54.85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-489","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":60.28,"normalizedScore":60.28,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-490","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":56.2,"normalizedScore":56.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-491","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":82.7,"normalizedScore":82.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-492","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":56.6,"normalizedScore":56.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-493","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":51.5,"normalizedScore":51.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-494","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":66.49,"normalizedScore":66.49,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-495","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":64.66,"normalizedScore":64.66,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-496","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":87.3,"normalizedScore":87.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-497","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":75.1,"normalizedScore":75.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-498","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":35.09,"normalizedScore":35.09,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-499","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":40.23,"normalizedScore":40.23,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-500","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":63.7,"normalizedScore":63.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-501","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":69.2,"normalizedScore":69.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-502","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":55.68,"normalizedScore":55.68,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-503","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":60.55,"normalizedScore":60.55,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-504","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":43.6,"normalizedScore":43.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-505","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":67.6,"normalizedScore":67.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-506","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":68.4,"normalizedScore":68.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-507","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":62.9,"normalizedScore":62.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-508","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":53.31,"normalizedScore":53.31,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-509","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":59.97,"normalizedScore":59.97,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-510","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":80.9,"normalizedScore":80.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-511","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":83.6,"normalizedScore":83.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-512","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":84.4,"normalizedScore":84.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-513","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":49.1,"normalizedScore":49.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-514","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":56.85,"normalizedScore":56.85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-515","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":51.03,"normalizedScore":51.03,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-516","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":65.4,"normalizedScore":65.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-517","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":76.9,"normalizedScore":76.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-518","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":91.8,"normalizedScore":91.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-519","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":47.4,"normalizedScore":47.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-520","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":55.56,"normalizedScore":55.56,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-521","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":58.72,"normalizedScore":58.72,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-522","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":67.5,"normalizedScore":67.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-523","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":80.9,"normalizedScore":80.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-524","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":59.3,"normalizedScore":59.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-525","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":58.7,"normalizedScore":58.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-526","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":35.38,"normalizedScore":35.38,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-527","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":44.74,"normalizedScore":44.74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-528","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":52.9,"normalizedScore":52.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-529","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":63.5,"normalizedScore":63.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-530","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":60.98,"normalizedScore":60.98,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-531","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":62.31,"normalizedScore":62.31,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-532","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":84.5,"normalizedScore":84.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-533","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":81.2,"normalizedScore":81.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-534","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":59.85,"normalizedScore":59.85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-538","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":67.5,"normalizedScore":67.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-539","modelSlug":"gemini-3-pro-deep-think","modelName":"Gemini 3 Pro Deep Think","providerId":"google","providerName":"Google","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":88.3,"normalizedScore":88.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-540","modelSlug":"gemini-3-pro-deep-think","modelName":"Gemini 3 Pro Deep Think","providerId":"google","providerName":"Google","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":83.6,"normalizedScore":83.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-541","modelSlug":"gemini-3-pro-deep-think","modelName":"Gemini 3 Pro Deep Think","providerId":"google","providerName":"Google","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":95,"normalizedScore":95,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-542","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":27.31,"normalizedScore":27.31,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-543","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":50.09,"normalizedScore":50.09,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-544","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":52.3,"normalizedScore":52.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-545","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":65.2,"normalizedScore":65.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-546","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":67.6,"normalizedScore":67.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-547","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":25.41,"normalizedScore":25.41,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-548","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":45.94,"normalizedScore":45.94,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-549","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":62.2,"normalizedScore":62.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-550","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":43.8,"normalizedScore":43.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-551","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":71.4,"normalizedScore":71.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-552","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":50.1,"normalizedScore":50.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-553","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"bl-agents","benchmarkName":"BenchLM Agentic prior","benchmarkCategory":"agents","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":52.21,"normalizedScore":52.21,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-554","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"bl-coding","benchmarkName":"BenchLM Coding prior","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":59.28,"normalizedScore":59.28,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-555","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":48.4,"normalizedScore":48.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-556","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"bl-mathematics","benchmarkName":"BenchLM Math prior","benchmarkCategory":"mathematics","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":24.8,"normalizedScore":24.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-557","modelSlug":"grok-4-1","modelName":"Grok 4.1","providerId":"xai","providerName":"xAI","benchmarkSlug":"bl-reasoning","benchmarkName":"BenchLM Reasoning prior","benchmarkCategory":"reasoning","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":90.8,"normalizedScore":90.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-558","modelSlug":"grok-4-1","modelName":"Grok 4.1","providerId":"xai","providerName":"xAI","benchmarkSlug":"bl-research","benchmarkName":"BenchLM Knowledge prior","benchmarkCategory":"research","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":90.1,"normalizedScore":90.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-559","modelSlug":"grok-4-1","modelName":"Grok 4.1","providerId":"xai","providerName":"xAI","benchmarkSlug":"bl-multimodal","benchmarkName":"BenchLM Multimodal prior","benchmarkCategory":"multimodal","benchmarkOrganisation":"BenchLM","benchmarkVersion":"bench-align-v5.1","score":93.1,"normalizedScore":93.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BenchLM public leaderboard / CursorBench snapshot for the exact model variant.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-leaderboard","sourceTitle":"BenchLM overall leaderboard snapshot","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"unverified"},{"resultId":"evidence-2026-07-560","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu","benchmarkName":"MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":null,"score":86,"normalizedScore":86,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-561","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"webarena","benchmarkName":"WebArena","benchmarkCategory":"agents","benchmarkOrganisation":"WebArena authors","benchmarkVersion":null,"score":69,"normalizedScore":69,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-562","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2","score":64.4,"normalizedScore":64.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-563","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2","score":62,"normalizedScore":62,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-564","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2","score":60.8,"normalizedScore":60.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-565","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":89.5,"normalizedScore":89.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-566","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":88.5,"normalizedScore":88.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-567","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":85.7,"normalizedScore":85.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-568","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":85.2,"normalizedScore":85.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-569","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":82,"normalizedScore":82,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-570","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":79.2,"normalizedScore":79.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-571","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":89.8,"normalizedScore":89.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-573","modelSlug":"gemini-3-pro-deep-think","modelName":"Gemini 3 Pro Deep Think","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2","score":45.1,"normalizedScore":45.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-574","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2","score":42.5,"normalizedScore":42.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-575","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2","score":31.1,"normalizedScore":31.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-576","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026-05","score":0.8,"normalizedScore":0.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-577","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":87.1,"normalizedScore":87.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-578","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":84.8,"normalizedScore":84.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-579","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":91.7,"normalizedScore":91.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-580","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":null,"score":66.3,"normalizedScore":66.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-581","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":null,"score":14.2,"normalizedScore":14.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-582","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":null,"score":13.9,"normalizedScore":13.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-583","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":null,"score":8.3,"normalizedScore":8.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-584","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":80.8,"normalizedScore":80.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-585","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":72.7,"normalizedScore":72.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-586","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":72.1,"normalizedScore":72.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-587","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":66.3,"normalizedScore":66.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-588","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":39,"normalizedScore":39,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-589","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":81,"normalizedScore":81,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-590","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":80,"normalizedScore":80,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-591","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":68.4,"normalizedScore":68.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-592","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":62.1,"normalizedScore":62.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-594","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":54.7,"normalizedScore":54.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-595","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":53,"normalizedScore":53,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-596","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":52.3,"normalizedScore":52.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-597","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":50.4,"normalizedScore":50.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-598","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":50.4,"normalizedScore":50.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-599","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":49,"normalizedScore":49,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-600","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":48,"normalizedScore":48,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-601","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":37.7,"normalizedScore":37.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-602","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":30.8,"normalizedScore":30.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-603","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":28.8,"normalizedScore":28.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-604","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":26.5,"normalizedScore":26.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-605","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":91.3,"normalizedScore":91.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-606","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":91.2,"normalizedScore":91.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-607","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":90.4,"normalizedScore":90.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-608","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":89.9,"normalizedScore":89.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-609","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":87,"normalizedScore":87,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-610","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":86,"normalizedScore":86,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-611","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":84.3,"normalizedScore":84.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-612","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":82.8,"normalizedScore":82.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-613","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":62.1,"normalizedScore":62.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-614","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":61.5,"normalizedScore":61.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-615","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":58.4,"normalizedScore":58.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-616","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":57.2,"normalizedScore":57.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-617","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":57.1,"normalizedScore":57.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-618","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":56.6,"normalizedScore":56.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-619","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":56.2,"normalizedScore":56.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-620","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":55.1,"normalizedScore":55.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-621","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":53.4,"normalizedScore":53.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-622","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":52.4,"normalizedScore":52.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-623","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":80.9,"normalizedScore":80.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-624","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":80.84,"normalizedScore":80.84,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-625","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":79.6,"normalizedScore":79.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-626","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":78.8,"normalizedScore":78.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-627","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":78,"normalizedScore":78,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-628","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":77.8,"normalizedScore":77.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-629","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":77.4,"normalizedScore":77.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-630","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":74.8,"normalizedScore":74.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-632","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":83.7,"normalizedScore":83.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-633","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":68,"normalizedScore":68,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-634","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":81,"normalizedScore":81,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-635","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":80.4,"normalizedScore":80.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-636","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":78.8,"normalizedScore":78.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-637","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":77.3,"normalizedScore":77.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-638","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":76.9,"normalizedScore":76.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-639","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":70.6,"normalizedScore":70.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-640","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":66.1,"normalizedScore":66.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-641","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":88.4,"normalizedScore":88.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-642","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":86.4,"normalizedScore":86.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-643","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":81.5,"normalizedScore":81.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-644","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":81.4,"normalizedScore":81.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-645","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":77.4,"normalizedScore":77.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-646","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":68.5,"normalizedScore":68.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-647","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":88.1,"normalizedScore":88.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-648","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":76.8,"normalizedScore":76.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-649","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":71.8,"normalizedScore":71.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-650","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":56.1,"normalizedScore":56.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-651","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":48.2,"normalizedScore":48.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-652","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":42.3,"normalizedScore":42.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-653","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":31.1,"normalizedScore":31.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-654","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":75.6,"normalizedScore":75.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-655","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":48.2,"normalizedScore":48.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-656","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":46.3,"normalizedScore":46.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-657","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":43.5,"normalizedScore":43.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-658","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":39.8,"normalizedScore":39.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-659","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":38,"normalizedScore":38,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-660","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":35.5,"normalizedScore":35.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-662","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":43.793,"normalizedScore":43.793,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-663","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":40.7,"normalizedScore":40.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-664","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":39,"normalizedScore":39,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-665","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":37.6,"normalizedScore":37.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-666","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":35.64,"normalizedScore":35.64,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-667","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":33.448,"normalizedScore":33.448,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-668","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":32.4,"normalizedScore":32.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-669","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":31.034,"normalizedScore":31.034,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-670","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":26.207,"normalizedScore":26.207,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-671","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":25.86,"normalizedScore":25.86,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-672","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":20.69,"normalizedScore":20.69,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-673","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":null,"score":16.434,"normalizedScore":16.434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-674","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":58.2,"normalizedScore":58.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-675","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":56.1,"normalizedScore":56.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-676","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":51.5,"normalizedScore":51.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-677","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":50.5,"normalizedScore":50.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-678","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":50.2,"normalizedScore":50.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-679","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":50.1,"normalizedScore":50.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-680","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":49.9,"normalizedScore":49.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-681","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":47,"normalizedScore":47,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-682","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":47,"normalizedScore":47,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-683","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":46.9,"normalizedScore":46.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-684","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":46.9,"normalizedScore":46.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-685","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":46.2,"normalizedScore":46.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-686","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":45.7,"normalizedScore":45.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-687","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":43.8,"normalizedScore":43.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-688","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":43.6,"normalizedScore":43.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-689","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":43.5,"normalizedScore":43.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-690","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":43.4,"normalizedScore":43.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-691","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":43.3,"normalizedScore":43.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-692","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":42.5,"normalizedScore":42.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-693","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.7,"normalizedScore":40.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-694","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":36.7,"normalizedScore":36.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-695","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":99.1,"normalizedScore":99.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-696","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":98.5,"normalizedScore":98.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-697","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":98.5,"normalizedScore":98.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-698","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":98.2,"normalizedScore":98.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-699","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":97.7,"normalizedScore":97.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-700","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":97.7,"normalizedScore":97.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-701","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":95,"normalizedScore":95,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-702","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":94.2,"normalizedScore":94.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-703","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":91.5,"normalizedScore":91.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-704","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":91.2,"normalizedScore":91.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-705","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":87.1,"normalizedScore":87.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-706","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":86.3,"normalizedScore":86.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-707","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":84.8,"normalizedScore":84.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-708","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":84.8,"normalizedScore":84.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-709","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":81.9,"normalizedScore":81.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-710","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":79.5,"normalizedScore":79.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-711","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":76,"normalizedScore":76,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-712","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":74,"normalizedScore":74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-713","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":59.9,"normalizedScore":59.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-714","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau-bench","benchmarkName":"τ-bench","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":null,"score":43.3,"normalizedScore":43.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-715","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":94.3,"normalizedScore":94.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-716","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":92.6,"normalizedScore":92.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-717","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":90.9,"normalizedScore":90.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-718","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":79.9,"normalizedScore":79.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-719","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":76.3,"normalizedScore":76.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-720","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":75.9,"normalizedScore":75.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-721","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":75.9,"normalizedScore":75.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-722","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":75.7,"normalizedScore":75.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-723","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":75.6,"normalizedScore":75.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-724","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":73.3,"normalizedScore":73.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-725","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":73.2,"normalizedScore":73.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-726","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":72.9,"normalizedScore":72.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-727","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":70.4,"normalizedScore":70.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-728","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":68.8,"normalizedScore":68.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-729","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":61.1,"normalizedScore":61.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-730","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":55.1,"normalizedScore":55.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-731","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":53.5,"normalizedScore":53.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-732","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":44.6,"normalizedScore":44.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-733","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":43.6,"normalizedScore":43.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-734","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifeval","benchmarkName":"IFEval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Google Research","benchmarkVersion":null,"score":41.2,"normalizedScore":41.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-735","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":70.5,"normalizedScore":70.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-736","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":67.2,"normalizedScore":67.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-737","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":66.7,"normalizedScore":66.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-738","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":64.9,"normalizedScore":64.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-739","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":62.3,"normalizedScore":62.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-740","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":61.5,"normalizedScore":61.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-741","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":61.1,"normalizedScore":61.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-742","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":58.4,"normalizedScore":58.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-743","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":55,"normalizedScore":55,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-744","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"cursor-bench","benchmarkName":"CursorBench","benchmarkCategory":"coding","benchmarkOrganisation":"Cursor","benchmarkVersion":"3.2","score":48.8,"normalizedScore":48.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-745","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":54.5,"normalizedScore":54.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-746","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":50.8,"normalizedScore":50.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-747","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":48.5,"normalizedScore":48.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-748","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":46.2,"normalizedScore":46.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-749","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":45.5,"normalizedScore":45.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-750","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":45.5,"normalizedScore":45.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-751","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":43.9,"normalizedScore":43.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-752","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":43.2,"normalizedScore":43.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-753","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":43.2,"normalizedScore":43.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-754","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":43.2,"normalizedScore":43.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-755","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":42.4,"normalizedScore":42.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-756","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":41.7,"normalizedScore":41.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-757","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":40.9,"normalizedScore":40.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-758","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":40.9,"normalizedScore":40.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-759","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":39.4,"normalizedScore":39.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-760","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":36.4,"normalizedScore":36.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-761","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":34.8,"normalizedScore":34.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-762","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":33.3,"normalizedScore":33.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-763","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":32.6,"normalizedScore":32.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-764","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"long-horizon-terminal-bench","benchmarkName":"Long-Horizon-Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Long-Horizon-Terminal-Bench authors","benchmarkVersion":"2026-07","score":31.8,"normalizedScore":31.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-765","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":96.1,"normalizedScore":96.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-766","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":99.2,"normalizedScore":99.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-767","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":95.8,"normalizedScore":95.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-768","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":95.8,"normalizedScore":95.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-769","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":95.3,"normalizedScore":95.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-770","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":95.3,"normalizedScore":95.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-771","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":95.1,"normalizedScore":95.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-772","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":99.8,"normalizedScore":99.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-773","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":96.7,"normalizedScore":96.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-774","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":85.71,"normalizedScore":85.71,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-775","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"super-gpqa","benchmarkName":"SuperGPQA","benchmarkCategory":"reasoning","benchmarkOrganisation":"SuperGPQA authors","benchmarkVersion":null,"score":95,"normalizedScore":95,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-776","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"super-gpqa","benchmarkName":"SuperGPQA","benchmarkCategory":"reasoning","benchmarkOrganisation":"SuperGPQA authors","benchmarkVersion":null,"score":95,"normalizedScore":95,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-777","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"super-gpqa","benchmarkName":"SuperGPQA","benchmarkCategory":"reasoning","benchmarkOrganisation":"SuperGPQA authors","benchmarkVersion":null,"score":73.6,"normalizedScore":73.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-778","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"super-gpqa","benchmarkName":"SuperGPQA","benchmarkCategory":"reasoning","benchmarkOrganisation":"SuperGPQA authors","benchmarkVersion":null,"score":71.6,"normalizedScore":71.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-779","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"super-gpqa","benchmarkName":"SuperGPQA","benchmarkCategory":"reasoning","benchmarkOrganisation":"SuperGPQA authors","benchmarkVersion":null,"score":71.4,"normalizedScore":71.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-780","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"super-gpqa","benchmarkName":"SuperGPQA","benchmarkCategory":"reasoning","benchmarkOrganisation":"SuperGPQA authors","benchmarkVersion":null,"score":70.6,"normalizedScore":70.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-781","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"super-gpqa","benchmarkName":"SuperGPQA","benchmarkCategory":"reasoning","benchmarkOrganisation":"SuperGPQA authors","benchmarkVersion":null,"score":69.2,"normalizedScore":69.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-782","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"super-gpqa","benchmarkName":"SuperGPQA","benchmarkCategory":"reasoning","benchmarkOrganisation":"SuperGPQA authors","benchmarkVersion":null,"score":66.8,"normalizedScore":66.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-783","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":null,"score":65.3,"normalizedScore":65.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-784","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":null,"score":62.8,"normalizedScore":62.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-785","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":null,"score":62.7,"normalizedScore":62.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-786","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":null,"score":60.7,"normalizedScore":60.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-787","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":null,"score":58.5,"normalizedScore":58.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-788","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":null,"score":58.2,"normalizedScore":58.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-789","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":null,"score":51.9,"normalizedScore":51.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-790","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":null,"score":41.6,"normalizedScore":41.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-791","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":51.4,"normalizedScore":51.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-792","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":51.2,"normalizedScore":51.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-793","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":48.6,"normalizedScore":48.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-794","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":48.5,"normalizedScore":48.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-795","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":45.6,"normalizedScore":45.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-796","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":42.8,"normalizedScore":42.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-797","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":42.6,"normalizedScore":42.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-798","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":42.2,"normalizedScore":42.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-799","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":42.1,"normalizedScore":42.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-800","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":39.2,"normalizedScore":39.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-801","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":37.5,"normalizedScore":37.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-802","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":27.8,"normalizedScore":27.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-803","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":25.6,"normalizedScore":25.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-804","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"multi-swe-bench","benchmarkName":"Multi-SWE-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"Multi-SWE-Bench","benchmarkVersion":null,"score":52.7,"normalizedScore":52.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-805","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":81.3,"normalizedScore":81.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-806","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":80.5,"normalizedScore":80.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-807","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":79.1,"normalizedScore":79.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-808","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":76.3,"normalizedScore":76.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-809","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":75.8,"normalizedScore":75.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-810","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":58,"normalizedScore":58,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-811","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":82.9,"normalizedScore":82.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-812","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":79.9,"normalizedScore":79.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-813","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":77.2,"normalizedScore":77.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-814","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":77.1,"normalizedScore":77.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-815","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":76.3,"normalizedScore":76.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-816","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":75.9,"normalizedScore":75.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-817","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":75.9,"normalizedScore":75.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-818","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":75.9,"normalizedScore":75.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-819","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":75.7,"normalizedScore":75.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-820","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":75.6,"normalizedScore":75.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-821","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":75.4,"normalizedScore":75.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-822","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":73.9,"normalizedScore":73.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-823","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":73.3,"normalizedScore":73.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-824","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":73.2,"normalizedScore":73.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-825","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":72.9,"normalizedScore":72.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-826","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":72.7,"normalizedScore":72.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-827","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":72.3,"normalizedScore":72.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-828","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":71.2,"normalizedScore":71.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-829","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":70.4,"normalizedScore":70.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-830","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":70.2,"normalizedScore":70.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-831","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":68.8,"normalizedScore":68.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-832","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":63.5,"normalizedScore":63.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-833","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":62.2,"normalizedScore":62.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-834","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":61.1,"normalizedScore":61.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-835","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":55.1,"normalizedScore":55.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-836","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":53.5,"normalizedScore":53.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-837","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":48.2,"normalizedScore":48.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-838","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":44.6,"normalizedScore":44.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-839","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":43.6,"normalizedScore":43.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-840","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":41.2,"normalizedScore":41.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-841","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":36.2,"normalizedScore":36.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-842","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"hmmt-feb-2026","benchmarkName":"HMMT Feb 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"HMMT","benchmarkVersion":null,"score":97.1,"normalizedScore":97.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-843","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"hmmt-feb-2026","benchmarkName":"HMMT Feb 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"HMMT","benchmarkVersion":null,"score":92.9,"normalizedScore":92.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-844","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"hmmt-feb-2026","benchmarkName":"HMMT Feb 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"HMMT","benchmarkVersion":null,"score":92.5,"normalizedScore":92.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-845","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"hmmt-feb-2026","benchmarkName":"HMMT Feb 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"HMMT","benchmarkVersion":null,"score":87.8,"normalizedScore":87.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-846","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"hmmt-feb-2026","benchmarkName":"HMMT Feb 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"HMMT","benchmarkVersion":null,"score":87.1,"normalizedScore":87.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-847","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"hmmt-feb-2026","benchmarkName":"HMMT Feb 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"HMMT","benchmarkVersion":null,"score":86.4,"normalizedScore":86.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-848","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"hmmt-feb-2026","benchmarkName":"HMMT Feb 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"HMMT","benchmarkVersion":null,"score":85.3,"normalizedScore":85.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-849","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"hmmt-feb-2026","benchmarkName":"HMMT Feb 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"HMMT","benchmarkVersion":null,"score":82.6,"normalizedScore":82.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-850","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"hmmt-feb-2026","benchmarkName":"HMMT Feb 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"HMMT","benchmarkVersion":null,"score":40.8,"normalizedScore":40.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-851","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"hmmt-feb-2026","benchmarkName":"HMMT Feb 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"HMMT","benchmarkVersion":null,"score":31.7,"normalizedScore":31.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public benchmark leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-852","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":80.71,"normalizedScore":80.71,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-853","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":80.28,"normalizedScore":80.28,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-854","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":79.93,"normalizedScore":79.93,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-855","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":78.31,"normalizedScore":78.31,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-856","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":77.22,"normalizedScore":77.22,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-857","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":76.91,"normalizedScore":76.91,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-858","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":76.33,"normalizedScore":76.33,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-859","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":75.96,"normalizedScore":75.96,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-860","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":75.47,"normalizedScore":75.47,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-861","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":75.02,"normalizedScore":75.02,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-862","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":74.29,"normalizedScore":74.29,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-863","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":72.76,"normalizedScore":72.76,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-864","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":70.85,"normalizedScore":70.85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-865","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":70.18,"normalizedScore":70.18,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-866","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":70.13,"normalizedScore":70.13,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-867","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":70.02,"normalizedScore":70.02,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-868","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"livebench","benchmarkName":"LiveBench","benchmarkCategory":"reasoning","benchmarkOrganisation":"LiveBench team","benchmarkVersion":null,"score":69.07,"normalizedScore":69.07,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-livebench","sourceTitle":"LiveBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/livebench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-869","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"paperbench","benchmarkName":"PaperBench","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":63.5,"normalizedScore":63.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-paperbench","sourceTitle":"PaperBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/paperbench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-870","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"paperbench","benchmarkName":"PaperBench","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":52.6,"normalizedScore":52.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-paperbench","sourceTitle":"PaperBench scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/paperbench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-871","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"arena-hard-auto","benchmarkName":"Arena-Hard-Auto","benchmarkCategory":"instruction-following","benchmarkOrganisation":"LMSYS Org","benchmarkVersion":null,"score":55.1,"normalizedScore":55.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"llm-stats-arena-hard","sourceTitle":"Arena-Hard scores via LLM Stats","sourcePublisher":"LLM Stats","sourceUrl":"https://llm-stats.com/benchmarks/arena-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-872","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":52.3,"normalizedScore":52.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-873","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":52.3,"normalizedScore":52.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-874","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":48.2,"normalizedScore":48.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-875","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":47.8,"normalizedScore":47.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-876","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":46.1,"normalizedScore":46.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-877","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":45.5,"normalizedScore":45.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-878","modelSlug":"grok-4-1","modelName":"Grok 4.1","providerId":"xai","providerName":"xAI","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":39.7,"normalizedScore":39.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-879","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":38.5,"normalizedScore":38.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-880","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":37.4,"normalizedScore":37.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-881","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":36.5,"normalizedScore":36.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-882","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":35.2,"normalizedScore":35.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-883","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":31.5,"normalizedScore":31.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-884","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gaia","benchmarkName":"GAIA","benchmarkCategory":"agents","benchmarkOrganisation":"GAIA authors","benchmarkVersion":null,"score":30.8,"normalizedScore":30.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-gaia","sourceTitle":"GAIA leaderboard via BenchLM","sourcePublisher":"BenchLM / GAIA","sourceUrl":"https://benchlm.ai/benchmarks/gaia","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-885","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitbench","benchmarkName":"ExploitBench","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitBench authors","benchmarkVersion":"2","score":34,"normalizedScore":34,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-exploitbench","sourceTitle":"ExploitBench via BenchLM","sourcePublisher":"BenchLM / ExploitBench","sourceUrl":"https://benchlm.ai/benchmarks/exploitBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-886","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"exploitbench","benchmarkName":"ExploitBench","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitBench authors","benchmarkVersion":"2","score":27,"normalizedScore":27,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-exploitbench","sourceTitle":"ExploitBench via BenchLM","sourcePublisher":"BenchLM / ExploitBench","sourceUrl":"https://benchlm.ai/benchmarks/exploitBench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-887","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":88.1,"normalizedScore":88.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-888","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":83.6,"normalizedScore":83.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-889","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":82.2,"normalizedScore":82.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-890","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":76.8,"normalizedScore":76.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-891","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":76.4,"normalizedScore":76.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-892","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":75.3,"normalizedScore":75.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-893","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":74.2,"normalizedScore":74.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-894","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":73.2,"normalizedScore":73.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-895","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":71.8,"normalizedScore":71.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-896","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":70.6,"normalizedScore":70.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-897","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":69.4,"normalizedScore":69.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-898","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":64,"normalizedScore":64,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-899","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":56.1,"normalizedScore":56.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-900","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":48.2,"normalizedScore":48.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-901","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":42.3,"normalizedScore":42.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-902","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":31.1,"normalizedScore":31.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-903","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":29.5,"normalizedScore":29.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as published on the source leaderboard page.","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"benchlm-public-benchmarks","sourceTitle":"BenchLM public benchmark leaderboards","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-904","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":62.98,"normalizedScore":62.98,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-905","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":70,"normalizedScore":70,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-906","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":40.15,"normalizedScore":40.15,"unit":"index","scoreDirection":"higher","modelConfiguration":"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-907","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":28.5714,"normalizedScore":28.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-908","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":98.538,"normalizedScore":98.538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-909","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":26.8041,"normalizedScore":26.8041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-910","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":62.8788,"normalizedScore":62.8788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-911","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":14.1667,"normalizedScore":14.1667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-912","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":51.1191,"normalizedScore":51.1191,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Fable 5 (Adaptive Reasoning, Max Effort, Opus 4.8 Fallback)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-913","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":42.442,"normalizedScore":42.442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.5 Flash (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-914","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":69.3333,"normalizedScore":69.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.5 Flash (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-915","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":22.6833,"normalizedScore":22.6833,"unit":"index","scoreDirection":"higher","modelConfiguration":"Gemini 3.5 Flash (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-916","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":13.1429,"normalizedScore":13.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.5 Flash (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-917","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":95.3216,"normalizedScore":95.3216,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.5 Flash (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-918","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":25.3608,"normalizedScore":25.3608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.5 Flash (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-919","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":40.9091,"normalizedScore":40.9091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.5 Flash (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-920","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":47.0501,"normalizedScore":47.0501,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.5 Flash (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-921","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":40.3484,"normalizedScore":40.3484,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.5 Flash (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-922","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":1.6667,"normalizedScore":1.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.5 Flash (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-923","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":50.0746,"normalizedScore":50.0746,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.5 Flash (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-924","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":38.27,"normalizedScore":38.27,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5-Pro","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-925","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":73.3333,"normalizedScore":73.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5-Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-926","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":3.6,"normalizedScore":3.6,"unit":"index","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5-Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-927","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":4,"normalizedScore":4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5-Pro","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-928","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":94.152,"normalizedScore":94.152,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5-Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-929","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":8.6598,"normalizedScore":8.6598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5-Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-930","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":43.1818,"normalizedScore":43.1818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5-Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-931","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":2.4336,"normalizedScore":2.4336,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5-Pro","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-932","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":38.2298,"normalizedScore":38.2298,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5-Pro","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-933","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5-Pro","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-934","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":54.5905,"normalizedScore":54.5905,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Luna (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-935","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":74,"normalizedScore":74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Luna (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-936","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-11.2333,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Luna (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-937","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":20.5714,"normalizedScore":20.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Luna (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-938","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":27.2165,"normalizedScore":27.2165,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Luna (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-939","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":40.3202,"normalizedScore":40.3202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Luna (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-940","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":5,"normalizedScore":5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Luna (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-941","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":62.391,"normalizedScore":62.391,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Sol (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-942","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":73.6667,"normalizedScore":73.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Sol (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-943","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":21.7,"normalizedScore":21.7,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Sol (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-944","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":32.2857,"normalizedScore":32.2857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Sol (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-945","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":85.0877,"normalizedScore":85.0877,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Sol (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-946","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":32.9897,"normalizedScore":32.9897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Sol (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-947","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":65.9091,"normalizedScore":65.9091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Sol (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-948","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":56.2147,"normalizedScore":56.2147,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Sol (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-949","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":4.3995,"normalizedScore":4.3995,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Small 4 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-950","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":44.6667,"normalizedScore":44.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Small 4 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-951","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-29.9,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Mistral Small 4 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-952","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.2857,"normalizedScore":0.2857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Small 4 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-953","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":41.2281,"normalizedScore":41.2281,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Small 4 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-954","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":5.1546,"normalizedScore":5.1546,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Small 4 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-955","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":17.4242,"normalizedScore":17.4242,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Small 4 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-956","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":55.336,"normalizedScore":55.336,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-957","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":70.6667,"normalizedScore":70.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-958","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":15.3167,"normalizedScore":15.3167,"unit":"index","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-959","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":16.8571,"normalizedScore":16.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-960","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":28.2474,"normalizedScore":28.2474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-961","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":5,"normalizedScore":5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-962","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":44.6732,"normalizedScore":44.6732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 5 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-963","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":60.6667,"normalizedScore":60.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5-Turbo","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-964","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-15.0833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GLM-5-Turbo","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-965","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.2857,"normalizedScore":0.2857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5-Turbo","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-966","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":98.538,"normalizedScore":98.538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5-Turbo","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-967","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":33.3333,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5-Turbo","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-968","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":49.686,"normalizedScore":49.686,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.5 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-969","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":74.3333,"normalizedScore":74.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.5 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-970","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":20.0667,"normalizedScore":20.0667,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5.5 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-971","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":27.1429,"normalizedScore":27.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.5 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-972","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":93.8596,"normalizedScore":93.8596,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.5 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-973","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":31.3402,"normalizedScore":31.3402,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.5 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-974","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":60.6061,"normalizedScore":60.6061,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.5 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-975","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":37.6844,"normalizedScore":37.6844,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.5 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-976","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":45.8098,"normalizedScore":45.8098,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.5 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-977","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":4.1667,"normalizedScore":4.1667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.5 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-978","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":46.6428,"normalizedScore":46.6428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.5 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-979","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":4.5075,"normalizedScore":4.5075,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM 5V Turbo (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-980","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":61,"normalizedScore":61,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM 5V Turbo (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-981","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-18.9833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GLM 5V Turbo (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-982","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.5714,"normalizedScore":0.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM 5V Turbo (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-983","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":98.538,"normalizedScore":98.538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM 5V Turbo (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-984","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":32.5758,"normalizedScore":32.5758,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM 5V Turbo (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-985","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":74,"normalizedScore":74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-986","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":13.2667,"normalizedScore":13.2667,"unit":"index","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-987","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":4.5714,"normalizedScore":4.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-988","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":89.4737,"normalizedScore":89.4737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-989","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":46.9697,"normalizedScore":46.9697,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-990","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":54.6475,"normalizedScore":54.6475,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Terra (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-991","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":74,"normalizedScore":74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Terra (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-992","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-0.2167,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Terra (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-993","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":30,"normalizedScore":30,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Terra (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-994","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":86.2573,"normalizedScore":86.2573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Terra (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-995","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":31.7526,"normalizedScore":31.7526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Terra (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-996","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":57.5758,"normalizedScore":57.5758,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Terra (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-997","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":51.0358,"normalizedScore":51.0358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Terra (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-998","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":2.5,"normalizedScore":2.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.6 Terra (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-999","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":44.7415,"normalizedScore":44.7415,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M3","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1000","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":74,"normalizedScore":74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1001","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":1.3667,"normalizedScore":1.3667,"unit":"index","scoreDirection":"higher","modelConfiguration":"MiniMax-M3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1002","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":3.7143,"normalizedScore":3.7143,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M3","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1003","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":88.8889,"normalizedScore":88.8889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1004","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":12.9897,"normalizedScore":12.9897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1005","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":42.4242,"normalizedScore":42.4242,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1006","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":6.6667,"normalizedScore":6.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M3","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1007","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":32.1098,"normalizedScore":32.1098,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M3","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1008","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":43.858,"normalizedScore":43.858,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1009","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":70.6667,"normalizedScore":70.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1010","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":12.3667,"normalizedScore":12.3667,"unit":"index","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1011","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":3.1429,"normalizedScore":3.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1012","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":75.731,"normalizedScore":75.731,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1013","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":30.5155,"normalizedScore":30.5155,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1014","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":53.0303,"normalizedScore":53.0303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1015","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":28.0236,"normalizedScore":28.0236,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1016","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":39.8023,"normalizedScore":39.8023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1017","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":4.1667,"normalizedScore":4.1667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Sonnet 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1018","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":49.995,"normalizedScore":49.995,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1019","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":70.3333,"normalizedScore":70.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1020","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":26.1667,"normalizedScore":26.1667,"unit":"index","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1021","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":12,"normalizedScore":12,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1022","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":88.5965,"normalizedScore":88.5965,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1023","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":28.866,"normalizedScore":28.866,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1024","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":51.5152,"normalizedScore":51.5152,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1025","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":46.6573,"normalizedScore":46.6573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.7 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1026","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":63.3333,"normalizedScore":63.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1027","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":2,"normalizedScore":2,"unit":"index","scoreDirection":"higher","modelConfiguration":"GLM-5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1028","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":2,"normalizedScore":2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1029","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":98.2456,"normalizedScore":98.2456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1030","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":43.1818,"normalizedScore":43.1818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1031","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":14.4543,"normalizedScore":14.4543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1032","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":70.6667,"normalizedScore":70.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1033","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":13.5,"normalizedScore":13.5,"unit":"index","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1034","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":12.5714,"normalizedScore":12.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1035","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":92.1053,"normalizedScore":92.1053,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1036","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":46.2121,"normalizedScore":46.2121,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1037","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":33.0383,"normalizedScore":33.0383,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.6 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1038","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":6.6255,"normalizedScore":6.6255,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Large 3","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1039","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":34.6667,"normalizedScore":34.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Large 3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1040","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-39.4333,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Mistral Large 3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1041","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Large 3","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1042","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":24.5614,"normalizedScore":24.5614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Large 3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1043","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":5.7732,"normalizedScore":5.7732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Large 3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1044","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":15.9091,"normalizedScore":15.9091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Large 3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1045","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":44.734,"normalizedScore":44.734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1046","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":74,"normalizedScore":74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1047","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":5.65,"normalizedScore":5.65,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5.4 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1048","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":23.4286,"normalizedScore":23.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1049","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":87.1345,"normalizedScore":87.1345,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1050","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":30.3093,"normalizedScore":30.3093,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1051","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":57.5758,"normalizedScore":57.5758,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1052","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":33.2596,"normalizedScore":33.2596,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1053","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":32.8955,"normalizedScore":32.8955,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.7","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1054","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":68.6667,"normalizedScore":68.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.7","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1055","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":0.6833,"normalizedScore":0.6833,"unit":"index","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.7","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1056","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.5714,"normalizedScore":0.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.7","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1057","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":84.7953,"normalizedScore":84.7953,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.7","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1058","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":8.866,"normalizedScore":8.866,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.7","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1059","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":39.3939,"normalizedScore":39.3939,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.7","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1060","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":10.6195,"normalizedScore":10.6195,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.7","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1061","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":26.4595,"normalizedScore":26.4595,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.7","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1062","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":24.3725,"normalizedScore":24.3725,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.1 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1063","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":75,"normalizedScore":75,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.1 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1064","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":5.55,"normalizedScore":5.55,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5.1 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1065","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":4.8571,"normalizedScore":4.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.1 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1066","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":81.8713,"normalizedScore":81.8713,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.1 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1067","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":14.0206,"normalizedScore":14.0206,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.1 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1068","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":45.4545,"normalizedScore":45.4545,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.1 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1069","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":7.0785,"normalizedScore":7.0785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Flash-Lite","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1070","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":65.3333,"normalizedScore":65.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Flash-Lite","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1071","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-15.5167,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Flash-Lite","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1072","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.1429,"normalizedScore":1.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Flash-Lite","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1073","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":31.2865,"normalizedScore":31.2865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Flash-Lite","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1074","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":8.6598,"normalizedScore":8.6598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Flash-Lite","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1075","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":24.2424,"normalizedScore":24.2424,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Flash-Lite","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1076","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":12.1681,"normalizedScore":12.1681,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Flash-Lite","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1077","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Flash-Lite","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1078","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":28.0215,"normalizedScore":28.0215,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Flash-Lite","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1079","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":74,"normalizedScore":74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.3 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1080","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":9.8833,"normalizedScore":9.8833,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5.3 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1081","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":16.8571,"normalizedScore":16.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.3 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1082","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":85.9649,"normalizedScore":85.9649,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.3 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1083","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":53.0303,"normalizedScore":53.0303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.3 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1084","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":37.834,"normalizedScore":37.834,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.1 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1085","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":62.3333,"normalizedScore":62.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.1 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1086","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":1.9333,"normalizedScore":1.9333,"unit":"index","scoreDirection":"higher","modelConfiguration":"GLM-5.1 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1087","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":4.5714,"normalizedScore":4.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.1 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1088","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":97.6608,"normalizedScore":97.6608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.1 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1089","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":11.5464,"normalizedScore":11.5464,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.1 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1090","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":43.1818,"normalizedScore":43.1818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.1 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1091","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":40.2542,"normalizedScore":40.2542,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.1 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1092","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":60.6667,"normalizedScore":60.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1093","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":4.9167,"normalizedScore":4.9167,"unit":"index","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1094","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.2857,"normalizedScore":0.2857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Pro","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1095","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":95.0292,"normalizedScore":95.0292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1096","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":40.9091,"normalizedScore":40.9091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1097","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":40.3605,"normalizedScore":40.3605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Pro (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1098","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66.3333,"normalizedScore":66.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Pro (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1099","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-10.0167,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Pro (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1100","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":12.8571,"normalizedScore":12.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Pro (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1101","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":96.1988,"normalizedScore":96.1988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Pro (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1102","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":25.7732,"normalizedScore":25.7732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Pro (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1103","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":46.2121,"normalizedScore":46.2121,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Pro (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1104","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":24.2625,"normalizedScore":24.2625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Pro (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1105","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":38.3239,"normalizedScore":38.3239,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Pro (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1106","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":3.3333,"normalizedScore":3.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Pro (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1107","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":40.4357,"normalizedScore":40.4357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Pro (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1108","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":29.245,"normalizedScore":29.245,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.3 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1109","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":64.3333,"normalizedScore":64.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.3 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1110","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":18.3167,"normalizedScore":18.3167,"unit":"index","scoreDirection":"higher","modelConfiguration":"Grok 4.3 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1111","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":8,"normalizedScore":8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.3 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1112","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":97.6608,"normalizedScore":97.6608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.3 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1113","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":12.1649,"normalizedScore":12.1649,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.3 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1114","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":37.8788,"normalizedScore":37.8788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.3 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1115","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":17.0354,"normalizedScore":17.0354,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.3 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1116","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":32.7213,"normalizedScore":32.7213,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.3 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1117","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.3 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1118","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":26.2608,"normalizedScore":26.2608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.3 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1119","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":38.6575,"normalizedScore":38.6575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Max","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1120","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":69,"normalizedScore":69,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Max","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1121","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":14.0833,"normalizedScore":14.0833,"unit":"index","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Max","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1122","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":13.4286,"normalizedScore":13.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Max","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1123","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":94.7368,"normalizedScore":94.7368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Max","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1124","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":10.9278,"normalizedScore":10.9278,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Max","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1125","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":50.7576,"normalizedScore":50.7576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Max","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1126","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":42.467,"normalizedScore":42.467,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Max","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1127","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Max","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1128","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":45.0015,"normalizedScore":45.0015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Max","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1129","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66.3333,"normalizedScore":66.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3 Flash Preview (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1130","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":11.5667,"normalizedScore":11.5667,"unit":"index","scoreDirection":"higher","modelConfiguration":"Gemini 3 Flash Preview (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1131","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":8.5714,"normalizedScore":8.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3 Flash Preview (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1132","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":80.4094,"normalizedScore":80.4094,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3 Flash Preview (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1133","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":17.5258,"normalizedScore":17.5258,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3 Flash Preview (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1134","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":38.6364,"normalizedScore":38.6364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3 Flash Preview (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1135","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":27.7286,"normalizedScore":27.7286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3 Flash Preview (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1136","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":51.7315,"normalizedScore":51.7315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1137","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":67.6667,"normalizedScore":67.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1138","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":26.3833,"normalizedScore":26.3833,"unit":"index","scoreDirection":"higher","modelConfiguration":"Grok 4.5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1139","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":15.4286,"normalizedScore":15.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1140","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":32.5773,"normalizedScore":32.5773,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1141","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":13.3333,"normalizedScore":13.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1142","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":40.8236,"normalizedScore":40.8236,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1143","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":29.9945,"normalizedScore":29.9945,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 nano (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1144","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66,"normalizedScore":66,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 nano (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1145","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-29.55,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5.4 nano (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1146","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":9.2531,"normalizedScore":9.2531,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 nano (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1147","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":76.0234,"normalizedScore":76.0234,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 nano (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1148","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":21.0309,"normalizedScore":21.0309,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 nano (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1149","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":42.4242,"normalizedScore":42.4242,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 nano (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1150","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":24.9263,"normalizedScore":24.9263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 nano (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1151","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":24.3879,"normalizedScore":24.3879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 nano (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1152","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 nano (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1153","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":31.4831,"normalizedScore":31.4831,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 nano (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1154","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":34.427,"normalizedScore":34.427,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Flash (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1155","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":63,"normalizedScore":63,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Flash (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1156","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-22.9,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Flash (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1157","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":7.1429,"normalizedScore":7.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Flash (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1158","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":95.0292,"normalizedScore":95.0292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Flash (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1159","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":22.8866,"normalizedScore":22.8866,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Flash (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1160","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":35.6061,"normalizedScore":35.6061,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Flash (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1161","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":31.516,"normalizedScore":31.516,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Flash (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1162","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":1.6667,"normalizedScore":1.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Flash (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1163","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":39.5703,"normalizedScore":39.5703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V4 Flash (Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1164","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Omni","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1165","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66.6667,"normalizedScore":66.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Omni","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1166","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-17.4333,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Omni","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1167","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.1429,"normalizedScore":1.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Omni","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1168","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":91.2281,"normalizedScore":91.2281,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Omni","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1169","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":34.8485,"normalizedScore":34.8485,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Omni","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1170","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":55.006,"normalizedScore":55.006,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1171","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":67.6667,"normalizedScore":67.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1172","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":27.4333,"normalizedScore":27.4333,"unit":"index","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1173","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":20.8571,"normalizedScore":20.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1174","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":94.4444,"normalizedScore":94.4444,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1175","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":27.6289,"normalizedScore":27.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1176","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":58.3333,"normalizedScore":58.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1177","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":7.5,"normalizedScore":7.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1178","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":43.957,"normalizedScore":43.957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.8 (Adaptive Reasoning, Max Effort)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1179","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":25.4265,"normalizedScore":25.4265,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1180","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":65.3333,"normalizedScore":65.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1181","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-8.1167,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Kimi K2.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1182","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":3.1429,"normalizedScore":3.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1183","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":95.9064,"normalizedScore":95.9064,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1184","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":14.2268,"normalizedScore":14.2268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1185","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":34.8485,"normalizedScore":34.8485,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1186","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":11.5044,"normalizedScore":11.5044,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.5 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1187","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":20.3305,"normalizedScore":20.3305,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Haiku (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1188","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":70.3333,"normalizedScore":70.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Haiku (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1189","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-4.2167,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Haiku (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1190","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Haiku (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1191","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":54.6784,"normalizedScore":54.6784,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Haiku (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1192","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":9.0722,"normalizedScore":9.0722,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Haiku (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1193","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":27.2727,"normalizedScore":27.2727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Haiku (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1194","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":27.307,"normalizedScore":27.307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Haiku (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1195","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Haiku (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1196","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":29.3942,"normalizedScore":29.3942,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Haiku (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1197","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":21.813,"normalizedScore":21.813,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Plus","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1198","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":65,"normalizedScore":65,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1199","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":2.3667,"normalizedScore":2.3667,"unit":"index","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1200","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":9.1429,"normalizedScore":9.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Plus","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1201","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":92.9825,"normalizedScore":92.9825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1202","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":17.8694,"normalizedScore":17.8694,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1203","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":46.9697,"normalizedScore":46.9697,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1204","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":22.4189,"normalizedScore":22.4189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Plus","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1205","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":1.6667,"normalizedScore":1.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Plus","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1206","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":40.5551,"normalizedScore":40.5551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.7 Plus","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1207","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":70.6667,"normalizedScore":70.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3 Pro Preview (high)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1208","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":15.8,"normalizedScore":15.8,"unit":"index","scoreDirection":"higher","modelConfiguration":"Gemini 3 Pro Preview (high)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1209","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":9.1429,"normalizedScore":9.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3 Pro Preview (high)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1210","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":87.1345,"normalizedScore":87.1345,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3 Pro Preview (high)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1211","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":41.6667,"normalizedScore":41.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3 Pro Preview (high)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1212","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":23.1155,"normalizedScore":23.1155,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Pro Preview","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1213","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":72.6667,"normalizedScore":72.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Pro Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1214","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":32.9333,"normalizedScore":32.9333,"unit":"index","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Pro Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1215","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":17.7143,"normalizedScore":17.7143,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Pro Preview","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1216","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":95.614,"normalizedScore":95.614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Pro Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1217","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":16.4948,"normalizedScore":16.4948,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Pro Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1218","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":53.7879,"normalizedScore":53.7879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Pro Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1219","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":32.0059,"normalizedScore":32.0059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Pro Preview","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1220","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":30.3309,"normalizedScore":30.3309,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Pro Preview","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1221","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Pro Preview","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1222","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":42.1665,"normalizedScore":42.1665,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 3.1 Pro Preview","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1223","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":15.2015,"normalizedScore":15.2015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 31B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1224","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":62,"normalizedScore":62,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 31B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1225","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-45.4167,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Gemma 4 31B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1226","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.4286,"normalizedScore":1.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 31B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1227","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":59.9415,"normalizedScore":59.9415,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 31B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1228","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":15.0515,"normalizedScore":15.0515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 31B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1229","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":36.3636,"normalizedScore":36.3636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 31B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1230","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":37.2881,"normalizedScore":37.2881,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 31B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1231","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 31B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1232","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":28.3497,"normalizedScore":28.3497,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 31B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1233","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":43.7165,"normalizedScore":43.7165,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1234","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":63.3333,"normalizedScore":63.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1235","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":18,"normalizedScore":18,"unit":"index","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1236","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":15.1429,"normalizedScore":15.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1237","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":25.1546,"normalizedScore":25.1546,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1238","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":8.3333,"normalizedScore":8.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1239","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":21.4475,"normalizedScore":21.4475,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Medium 3.5","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1240","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":61,"normalizedScore":61,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Medium 3.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1241","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-36.3167,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Mistral Medium 3.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1242","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Medium 3.5","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1243","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":94.152,"normalizedScore":94.152,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Medium 3.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1244","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":14.433,"normalizedScore":14.433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Medium 3.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1245","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":33.3333,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Medium 3.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1246","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0.8333,"normalizedScore":0.8333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Medium 3.5","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1247","modelSlug":"mistral-medium-3-5","modelName":"Mistral Medium 3.5","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":33.7213,"normalizedScore":33.7213,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Mistral Medium 3.5","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1249","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":50.6945,"normalizedScore":50.6945,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.2 (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1250","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":71.3333,"normalizedScore":71.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.2 (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1251","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":3.9667,"normalizedScore":3.9667,"unit":"index","scoreDirection":"higher","modelConfiguration":"GLM-5.2 (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1252","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":20.8571,"normalizedScore":20.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.2 (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1253","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":99.1228,"normalizedScore":99.1228,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.2 (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1254","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":26.8041,"normalizedScore":26.8041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.2 (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1255","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":50.7576,"normalizedScore":50.7576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.2 (max)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1256","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":33.7021,"normalizedScore":33.7021,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.2 (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1257","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":42.6554,"normalizedScore":42.6554,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.2 (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1258","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":7.5,"normalizedScore":7.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.2 (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1259","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":42.7335,"normalizedScore":42.7335,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-5.2 (max)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1260","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":32.1845,"normalizedScore":32.1845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1261","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":69.6667,"normalizedScore":69.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1262","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":4.0833,"normalizedScore":4.0833,"unit":"index","scoreDirection":"higher","modelConfiguration":"Muse Spark","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1263","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":11.3306,"normalizedScore":11.3306,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1264","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":91.5205,"normalizedScore":91.5205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1265","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":19.5876,"normalizedScore":19.5876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1266","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":45.4545,"normalizedScore":45.4545,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1267","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":31.7695,"normalizedScore":31.7695,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Plus","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1268","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":69.6667,"normalizedScore":69.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1269","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":2.65,"normalizedScore":2.65,"unit":"index","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1270","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":2.8571,"normalizedScore":2.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Plus","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1271","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":97.6608,"normalizedScore":97.6608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1272","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":16.4948,"normalizedScore":16.4948,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1273","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":43.9394,"normalizedScore":43.9394,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1274","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybench","benchmarkName":"Cybench","benchmarkCategory":"agents","benchmarkOrganisation":"Stanford / Cybench authors","benchmarkVersion":null,"score":100,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Mythos Preview; unguided; 35-problem subset (system card via Cybench leaderboard note ⁶)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cybench-official","sourceTitle":"Cybench official leaderboard","sourcePublisher":"Cybench authors","sourceUrl":"https://cybench.github.io/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-1275","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybench","benchmarkName":"Cybench","benchmarkCategory":"agents","benchmarkOrganisation":"Stanford / Cybench authors","benchmarkVersion":null,"score":96,"normalizedScore":96,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.7; unguided; 35-problem subset (system card via Cybench leaderboard note ⁸)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cybench-official","sourceTitle":"Cybench official leaderboard","sourcePublisher":"Cybench authors","sourceUrl":"https://cybench.github.io/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-1276","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybench","benchmarkName":"Cybench","benchmarkCategory":"agents","benchmarkOrganisation":"Stanford / Cybench authors","benchmarkVersion":null,"score":93,"normalizedScore":93,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.6; unguided; 37-problem subset (system card via Cybench leaderboard note ⁵)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cybench-official","sourceTitle":"Cybench official leaderboard","sourcePublisher":"Cybench authors","sourceUrl":"https://cybench.github.io/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-1277","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybench","benchmarkName":"Cybench","benchmarkCategory":"agents","benchmarkOrganisation":"Stanford / Cybench authors","benchmarkVersion":null,"score":82,"normalizedScore":82,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Opus 4.5; unguided; 39-problem subset (system card via Cybench leaderboard note ³)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cybench-official","sourceTitle":"Cybench official leaderboard","sourcePublisher":"Cybench authors","sourceUrl":"https://cybench.github.io/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-1278","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"cybench","benchmarkName":"Cybench","benchmarkCategory":"agents","benchmarkOrganisation":"Stanford / Cybench authors","benchmarkVersion":null,"score":65.4,"normalizedScore":65.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark; unguided (Meta safety report via Cybench leaderboard note ⁷)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cybench-official","sourceTitle":"Cybench official leaderboard","sourcePublisher":"Cybench authors","sourceUrl":"https://cybench.github.io/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-1279","modelSlug":"grok-4-1","modelName":"Grok 4.1","providerId":"xai","providerName":"xAI","benchmarkSlug":"cybench","benchmarkName":"Cybench","benchmarkCategory":"agents","benchmarkOrganisation":"Stanford / Cybench authors","benchmarkVersion":null,"score":39,"normalizedScore":39,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.1 Thinking; unguided (xAI model card via Cybench leaderboard note ⁴)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"cybench-official","sourceTitle":"Cybench official leaderboard","sourcePublisher":"Cybench authors","sourceUrl":"https://cybench.github.io/","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-1280","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":33.54,"normalizedScore":33.54,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1281","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":69.3333,"normalizedScore":69.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1282","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-18.6833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1283","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":10,"normalizedScore":10,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1284","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":83.3333,"normalizedScore":83.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1285","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":21.4433,"normalizedScore":21.4433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1286","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":52.2727,"normalizedScore":52.2727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1287","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":59.176,"normalizedScore":59.176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1288","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":28.1711,"normalizedScore":28.1711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1289","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":35.2166,"normalizedScore":35.2166,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1290","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1291","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":34.6165,"normalizedScore":34.6165,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1292","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":49.8843,"normalizedScore":49.8843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1293","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":26.645,"normalizedScore":26.645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1294","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":87.4747,"normalizedScore":87.4747,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1295","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":73.2948,"normalizedScore":73.2948,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1296","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":73.2653,"normalizedScore":73.2653,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.4 mini (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1297","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":72.6667,"normalizedScore":72.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1298","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-1,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5.2 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1299","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":11.551,"normalizedScore":11.551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1300","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":84.7953,"normalizedScore":84.7953,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1301","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":46.9697,"normalizedScore":46.9697,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1302","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":52.0833,"normalizedScore":52.0833,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1303","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":35.4495,"normalizedScore":35.4495,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1304","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":90.303,"normalizedScore":90.303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1305","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":88.8889,"normalizedScore":88.8889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1306","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":99,"normalizedScore":99,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1307","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":75.4422,"normalizedScore":75.4422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1308","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":75.6667,"normalizedScore":75.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1309","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-2.4833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5.2 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1310","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":8.6694,"normalizedScore":8.6694,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1311","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":92.1053,"normalizedScore":92.1053,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1312","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":37.1212,"normalizedScore":37.1212,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1313","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":54.6296,"normalizedScore":54.6296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1314","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":33.4569,"normalizedScore":33.4569,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1315","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":89.899,"normalizedScore":89.899,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1316","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":76.3006,"normalizedScore":76.3006,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1317","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":77.619,"normalizedScore":77.619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5.2 Codex (xhigh)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1318","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":69,"normalizedScore":69,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1319","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-6.7667,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1320","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":5.1429,"normalizedScore":5.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1321","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":86.8421,"normalizedScore":86.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1322","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":37.8788,"normalizedScore":37.8788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1323","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.8565,"normalizedScore":40.8565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1324","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":25.5792,"normalizedScore":25.5792,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1325","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":83.7374,"normalizedScore":83.7374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1326","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":73.815,"normalizedScore":73.815,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1327","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":84.0212,"normalizedScore":84.0212,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1328","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":98.6667,"normalizedScore":98.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1329","modelSlug":"gpt-5-codex","modelName":"GPT-5 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":74.1497,"normalizedScore":74.1497,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 Codex (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1330","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":28.74,"normalizedScore":28.74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1331","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":75.6,"normalizedScore":75.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1332","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-8.0833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1333","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":5.7143,"normalizedScore":5.7143,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1334","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":84.7953,"normalizedScore":84.7953,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1335","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":19.5876,"normalizedScore":19.5876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1336","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":32.5758,"normalizedScore":32.5758,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1337","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":35.206,"normalizedScore":35.206,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1338","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":42.9398,"normalizedScore":42.9398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1339","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":26.506,"normalizedScore":26.506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1340","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":85.3535,"normalizedScore":85.3535,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1341","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":74.2197,"normalizedScore":74.2197,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1342","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":84.5503,"normalizedScore":84.5503,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1343","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":94.3333,"normalizedScore":94.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1344","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":73.0612,"normalizedScore":73.0612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1345","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66,"normalizedScore":66,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1346","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-10.7833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1347","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.4286,"normalizedScore":1.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1348","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":71.0526,"normalizedScore":71.0526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1349","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":28.7879,"normalizedScore":28.7879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1350","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.9722,"normalizedScore":40.9722,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1351","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":14.5968,"normalizedScore":14.5968,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1352","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":80.303,"normalizedScore":80.303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1353","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":68.8439,"normalizedScore":68.8439,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1354","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":69.2063,"normalizedScore":69.2063,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1355","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":85,"normalizedScore":85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1356","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":71.1565,"normalizedScore":71.1565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-5 mini (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1357","modelSlug":"o3-pro","modelName":"o3-pro","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":84.5455,"normalizedScore":84.5455,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3-pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1358","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":69.3333,"normalizedScore":69.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1359","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-15.2667,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1360","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.1429,"normalizedScore":1.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1361","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":80.7018,"normalizedScore":80.7018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1362","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":37.1212,"normalizedScore":37.1212,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1363","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.9722,"normalizedScore":40.9722,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1364","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":20.0447,"normalizedScore":20.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1365","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":82.7273,"normalizedScore":82.7273,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1366","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":70.0578,"normalizedScore":70.0578,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1367","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":80.8466,"normalizedScore":80.8466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1368","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":88.3333,"normalizedScore":88.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1369","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":71.4286,"normalizedScore":71.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o3","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1370","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":27.623,"normalizedScore":27.623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1371","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":65.6667,"normalizedScore":65.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1372","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":0.3333,"normalizedScore":0.3333,"unit":"index","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1373","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.1429,"normalizedScore":1.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1374","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":78.0702,"normalizedScore":78.0702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1375","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":18.9691,"normalizedScore":18.9691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1376","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":35.6061,"normalizedScore":35.6061,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1377","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":55.8052,"normalizedScore":55.8052,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1378","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":44.6759,"normalizedScore":44.6759,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1379","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":17.2845,"normalizedScore":17.2845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1380","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":83.4343,"normalizedScore":83.4343,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1381","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":68.7283,"normalizedScore":68.7283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1382","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":71.4286,"normalizedScore":71.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1383","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":88,"normalizedScore":88,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1384","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":57.2789,"normalizedScore":57.2789,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.5 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1385","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":15.2015,"normalizedScore":15.2015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1386","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66.3333,"normalizedScore":66.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1387","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1388","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":71.3661,"normalizedScore":71.3661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1389","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":34.3352,"normalizedScore":34.3352,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1390","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1391","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.8979,"normalizedScore":40.8979,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1392","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":11.8911,"normalizedScore":11.8911,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1393","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":80.9,"normalizedScore":80.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1394","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":67.9191,"normalizedScore":67.9191,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1395","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":65.3545,"normalizedScore":65.3545,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1396","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":80.3333,"normalizedScore":80.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1397","modelSlug":"claude-opus-4-1","modelName":"Claude Opus 4.1","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":55.4422,"normalizedScore":55.4422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4.1 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1398","modelSlug":"claude-opus-4","modelName":"Claude Opus 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":56.5145,"normalizedScore":56.5145,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1399","modelSlug":"claude-opus-4","modelName":"Claude Opus 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":33.6667,"normalizedScore":33.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1400","modelSlug":"claude-opus-4","modelName":"Claude Opus 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":73.3918,"normalizedScore":73.3918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1401","modelSlug":"claude-opus-4","modelName":"Claude Opus 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":31.0606,"normalizedScore":31.0606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1402","modelSlug":"claude-opus-4","modelName":"Claude Opus 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":39.8148,"normalizedScore":39.8148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1403","modelSlug":"claude-opus-4","modelName":"Claude Opus 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":11.6775,"normalizedScore":11.6775,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1404","modelSlug":"claude-opus-4","modelName":"Claude Opus 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":79.596,"normalizedScore":79.596,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1405","modelSlug":"claude-opus-4","modelName":"Claude Opus 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":63.5979,"normalizedScore":63.5979,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1406","modelSlug":"claude-opus-4","modelName":"Claude Opus 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":73.3333,"normalizedScore":73.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1407","modelSlug":"claude-opus-4","modelName":"Claude Opus 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":53.7415,"normalizedScore":53.7415,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Opus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1408","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":34.5195,"normalizedScore":34.5195,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1409","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":69.6667,"normalizedScore":69.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1410","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":6.4167,"normalizedScore":6.4167,"unit":"index","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1411","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":8,"normalizedScore":8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1412","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":95.9064,"normalizedScore":95.9064,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1413","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":20.6186,"normalizedScore":20.6186,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1414","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":43.9394,"normalizedScore":43.9394,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1415","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":65.9176,"normalizedScore":65.9176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1416","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":28.4661,"normalizedScore":28.4661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1417","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":31.1864,"normalizedScore":31.1864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1418","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1419","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":38.5258,"normalizedScore":38.5258,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1420","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":53.4722,"normalizedScore":53.4722,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1421","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":35.9129,"normalizedScore":35.9129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1422","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":91.1111,"normalizedScore":91.1111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1423","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":79.3642,"normalizedScore":79.3642,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1424","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":75.9864,"normalizedScore":75.9864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1425","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":19.5527,"normalizedScore":19.5527,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.6","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1426","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":34.3395,"normalizedScore":34.3395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1427","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66.3333,"normalizedScore":66.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1428","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-10.7,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1429","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":10,"normalizedScore":10,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1430","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":90.0585,"normalizedScore":90.0585,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1431","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":18.1443,"normalizedScore":18.1443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1432","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":44.697,"normalizedScore":44.697,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1433","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":67.4157,"normalizedScore":67.4157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1434","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0.8333,"normalizedScore":0.8333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1435","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":40.2268,"normalizedScore":40.2268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1436","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":47.4537,"normalizedScore":47.4537,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1437","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":32.7618,"normalizedScore":32.7618,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1438","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":89.596,"normalizedScore":89.596,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1439","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":63.1293,"normalizedScore":63.1293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1440","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":22.5286,"normalizedScore":22.5286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2.7 Code","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1441","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":58,"normalizedScore":58,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.20 0309 v2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1442","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":15.35,"normalizedScore":15.35,"unit":"index","scoreDirection":"higher","modelConfiguration":"Grok 4.20 0309 v2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1443","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":6.5714,"normalizedScore":6.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.20 0309 v2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1444","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":92.9825,"normalizedScore":92.9825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.20 0309 v2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1445","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":37.8788,"normalizedScore":37.8788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.20 0309 v2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1446","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":45.6019,"normalizedScore":45.6019,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.20 0309 v2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1447","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":32.2057,"normalizedScore":32.2057,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.20 0309 v2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1448","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":91.1111,"normalizedScore":91.1111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.20 0309 v2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1449","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":74.5665,"normalizedScore":74.5665,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.20 0309 v2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1450","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":81.2245,"normalizedScore":81.2245,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4.20 0309 v2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1451","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":31.986,"normalizedScore":31.986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1452","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":68.6667,"normalizedScore":68.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1453","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-19.7833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1454","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.1429,"normalizedScore":1.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1455","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":94.152,"normalizedScore":94.152,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1456","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":15.2577,"normalizedScore":15.2577,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1457","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":34.8485,"normalizedScore":34.8485,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1458","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":60.6742,"normalizedScore":60.6742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1459","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":39.8148,"normalizedScore":39.8148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1460","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":21.5941,"normalizedScore":21.5941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1461","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":84.2424,"normalizedScore":84.2424,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1462","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":74.6243,"normalizedScore":74.6243,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1463","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":67.551,"normalizedScore":67.551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1464","modelSlug":"qwen3-6-max","modelName":"Qwen3.6 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":69.6667,"normalizedScore":69.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Max Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1465","modelSlug":"qwen3-6-max","modelName":"Qwen3.6 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":10.2,"normalizedScore":10.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Max Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1466","modelSlug":"qwen3-6-max","modelName":"Qwen3.6 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":3.7061,"normalizedScore":3.7061,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Max Preview","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1467","modelSlug":"qwen3-6-max","modelName":"Qwen3.6 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":95.9064,"normalizedScore":95.9064,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Max Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1468","modelSlug":"qwen3-6-max","modelName":"Qwen3.6 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":43.9394,"normalizedScore":43.9394,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Max Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1469","modelSlug":"qwen3-6-max","modelName":"Qwen3.6 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":46.875,"normalizedScore":46.875,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Max Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1470","modelSlug":"qwen3-6-max","modelName":"Qwen3.6 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":28.8693,"normalizedScore":28.8693,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Max Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1471","modelSlug":"qwen3-6-max","modelName":"Qwen3.6 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":88.7879,"normalizedScore":88.7879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Max Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1472","modelSlug":"qwen3-6-max","modelName":"Qwen3.6 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":76.5986,"normalizedScore":76.5986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.6 Max Preview","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1473","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":23.0955,"normalizedScore":23.0955,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1474","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":65.6667,"normalizedScore":65.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1475","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-29.7833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1476","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.7061,"normalizedScore":1.7061,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1477","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":95.614,"normalizedScore":95.614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1478","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":13.4021,"normalizedScore":13.4021,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1479","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":40.9091,"normalizedScore":40.9091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1480","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":51.3109,"normalizedScore":51.3109,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1481","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":15.3392,"normalizedScore":15.3392,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1482","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":34.0866,"normalizedScore":34.0866,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1483","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1484","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":30.8863,"normalizedScore":30.8863,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1485","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":42.0139,"normalizedScore":42.0139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1486","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":27.2938,"normalizedScore":27.2938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1487","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":89.2929,"normalizedScore":89.2929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1488","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":77.2832,"normalizedScore":77.2832,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1489","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":78.7755,"normalizedScore":78.7755,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1490","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":12.1956,"normalizedScore":12.1956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 397B A17B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1491","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":23.901,"normalizedScore":23.901,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1492","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66.6667,"normalizedScore":66.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1493","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-39.5833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1494","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.5714,"normalizedScore":0.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1495","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":93.5673,"normalizedScore":93.5673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1496","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":13.6082,"normalizedScore":13.6082,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1497","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":31.0606,"normalizedScore":31.0606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1498","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":47.5655,"normalizedScore":47.5655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1499","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":42.0139,"normalizedScore":42.0139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1500","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":23.4476,"normalizedScore":23.4476,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1501","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":85.6566,"normalizedScore":85.6566,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1502","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":74.9711,"normalizedScore":74.9711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1503","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":75.7143,"normalizedScore":75.7143,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 122B A10B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1504","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":67.3333,"normalizedScore":67.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1505","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-42.0167,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Qwen3.5 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1506","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.8531,"normalizedScore":0.8531,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1507","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":93.8596,"normalizedScore":93.8596,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1508","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":32.5758,"normalizedScore":32.5758,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1509","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":35.4991,"normalizedScore":35.4991,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1510","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":39.4676,"normalizedScore":39.4676,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1511","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":22.1965,"normalizedScore":22.1965,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1512","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":85.7576,"normalizedScore":85.7576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1513","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":75.0289,"normalizedScore":75.0289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1514","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":75.5782,"normalizedScore":75.5782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 27B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1515","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":18.274,"normalizedScore":18.274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1516","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":65,"normalizedScore":65,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1517","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-20.8833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1518","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":2.8571,"normalizedScore":2.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1519","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":90.6433,"normalizedScore":90.6433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1520","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":18.7629,"normalizedScore":18.7629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1521","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":35.6061,"normalizedScore":35.6061,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1522","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":46.8165,"normalizedScore":46.8165,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1523","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":14.528,"normalizedScore":14.528,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1524","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":38.8889,"normalizedScore":38.8889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1525","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":22.2428,"normalizedScore":22.2428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1526","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":84.0404,"normalizedScore":84.0404,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1527","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":86.2434,"normalizedScore":86.2434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1528","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":92,"normalizedScore":92,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1529","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":60.6803,"normalizedScore":60.6803,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.2 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1530","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":19.464,"normalizedScore":19.464,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1531","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":65,"normalizedScore":65,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1532","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-24.3167,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1533","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.7143,"normalizedScore":1.7143,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1534","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":37.1345,"normalizedScore":37.1345,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1535","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":15.8763,"normalizedScore":15.8763,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1536","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":30.303,"normalizedScore":30.303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1537","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":44.9438,"normalizedScore":44.9438,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1538","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.625,"normalizedScore":40.625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1539","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":15.2456,"normalizedScore":15.2456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1540","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":79.1919,"normalizedScore":79.1919,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1541","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":79.7884,"normalizedScore":79.7884,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1542","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":89.6667,"normalizedScore":89.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1543","modelSlug":"deepseek-v3-1-terminus","modelName":"DeepSeek V3.1 Terminus","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":57.0068,"normalizedScore":57.0068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek V3.1 Terminus (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1544","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":33.265,"normalizedScore":33.265,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1545","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":64,"normalizedScore":64,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1546","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-34.6,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1547","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.7102,"normalizedScore":1.7102,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1548","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":95.9064,"normalizedScore":95.9064,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1549","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":9.6907,"normalizedScore":9.6907,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1550","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":31.8182,"normalizedScore":31.8182,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1551","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":45.3184,"normalizedScore":45.3184,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1552","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":45.1389,"normalizedScore":45.1389,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1553","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":25.1158,"normalizedScore":25.1158,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1554","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":85.8586,"normalizedScore":85.8586,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1555","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":89.418,"normalizedScore":89.418,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1556","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":95,"normalizedScore":95,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1557","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":67.8912,"normalizedScore":67.8912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.7 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1558","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66,"normalizedScore":66,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1559","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-39.7,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1560","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.1429,"normalizedScore":1.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.5","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1561","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":95.3216,"normalizedScore":95.3216,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1562","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":34.8485,"normalizedScore":34.8485,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1563","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":42.5926,"normalizedScore":42.5926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1564","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":19.1381,"normalizedScore":19.1381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1565","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":84.8485,"normalizedScore":84.8485,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1566","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":71.6327,"normalizedScore":71.6327,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1567","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":32.2505,"normalizedScore":32.2505,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1568","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":62.6667,"normalizedScore":62.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1569","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-9.3333,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1570","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":3.7143,"normalizedScore":3.7143,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1571","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":90.6433,"normalizedScore":90.6433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1572","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":6.5979,"normalizedScore":6.5979,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1573","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":41.6667,"normalizedScore":41.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1574","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":63.6704,"normalizedScore":63.6704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1575","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":43.0556,"normalizedScore":43.0556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1576","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":25.1622,"normalizedScore":25.1622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1577","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":84.9495,"normalizedScore":84.9495,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1578","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":75.4335,"normalizedScore":75.4335,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1579","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":67.1429,"normalizedScore":67.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2.5","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1580","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":40.3605,"normalizedScore":40.3605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1581","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":63,"normalizedScore":63,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1582","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-43.45,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1583","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":4.2857,"normalizedScore":4.2857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1584","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":95.0292,"normalizedScore":95.0292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1585","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":28.0303,"normalizedScore":28.0303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1586","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":3.3333,"normalizedScore":3.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1587","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":40.4357,"normalizedScore":40.4357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1588","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":39.3519,"normalizedScore":39.3519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1589","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":21.1307,"normalizedScore":21.1307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1590","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":84.6465,"normalizedScore":84.6465,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1591","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":86.7725,"normalizedScore":86.7725,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1592","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":96.3333,"normalizedScore":96.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1593","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":64.2177,"normalizedScore":64.2177,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1594","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":18.8277,"normalizedScore":18.8277,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiMo-V2-Flash (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1595","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":33.19,"normalizedScore":33.19,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1596","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":67,"normalizedScore":67,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1597","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-0.8333,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1598","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":3.1429,"normalizedScore":3.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1599","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":83.3333,"normalizedScore":83.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1600","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":13.8144,"normalizedScore":13.8144,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1601","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":36.3636,"normalizedScore":36.3636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1602","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":53.9326,"normalizedScore":53.9326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1603","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":3.3333,"normalizedScore":3.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-harvey-lab","sourceTitle":"Harvey LAB-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/harvey-lab-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1604","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":28.8869,"normalizedScore":28.8869,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-enterpriseops-gym","sourceTitle":"EnterpriseOps-Gym-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/enterprise-ops-gym-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1605","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":39.9306,"normalizedScore":39.9306,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1606","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":26.5524,"normalizedScore":26.5524,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1607","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":86.6667,"normalizedScore":86.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1608","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":81.3605,"normalizedScore":81.3605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1609","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":5.6706,"normalizedScore":5.6706,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nemotron 3 Ultra 550B A55B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1610","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":13.0665,"normalizedScore":13.0665,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1611","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":55.6667,"normalizedScore":55.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1612","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-48.0667,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1613","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1614","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":43.5673,"normalizedScore":43.5673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1615","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":11.7526,"normalizedScore":11.7526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1616","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":13.64,"normalizedScore":13.64,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1617","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":38.9513,"normalizedScore":38.9513,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1618","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":23.6347,"normalizedScore":23.6347,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1619","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.0463,"normalizedScore":40.0463,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1620","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":18.2576,"normalizedScore":18.2576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1621","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":79.1919,"normalizedScore":79.1919,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1622","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":69.2486,"normalizedScore":69.2486,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1623","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":72.449,"normalizedScore":72.449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 26B A4B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1624","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":10.71,"normalizedScore":10.71,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1625","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":46,"normalizedScore":46,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1626","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-3.9833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1627","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.2857,"normalizedScore":0.2857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1628","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":80.7018,"normalizedScore":80.7018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1629","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":5.7732,"normalizedScore":5.7732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1630","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":25,"normalizedScore":25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1631","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":22.8464,"normalizedScore":22.8464,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1632","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":37.8472,"normalizedScore":37.8472,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1633","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":11.3531,"normalizedScore":11.3531,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1634","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":76.0606,"normalizedScore":76.0606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1635","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":63.237,"normalizedScore":63.237,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1636","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":73.9456,"normalizedScore":73.9456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Command A+","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1637","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":8.811,"normalizedScore":8.811,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1638","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":54.3333,"normalizedScore":54.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1639","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-48.05,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1640","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1641","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":92.6901,"normalizedScore":92.6901,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1642","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":7.6289,"normalizedScore":7.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1643","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":24.2424,"normalizedScore":24.2424,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1644","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":29.588,"normalizedScore":29.588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1645","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":42.7083,"normalizedScore":42.7083,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1646","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":8.9435,"normalizedScore":8.9435,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1647","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":78.4848,"normalizedScore":78.4848,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1648","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":64.5087,"normalizedScore":64.5087,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1649","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":73.0159,"normalizedScore":73.0159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1650","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":89,"normalizedScore":89,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1651","modelSlug":"nova-2-pro","modelName":"Nova 2 Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":79.0476,"normalizedScore":79.0476,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Nova 2.0 Pro Preview (medium)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1652","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":68,"normalizedScore":68,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1653","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":3.75,"normalizedScore":3.75,"unit":"index","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1654","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":2,"normalizedScore":2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1655","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":74.8538,"normalizedScore":74.8538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1656","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":37.8788,"normalizedScore":37.8788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1657","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":45.7176,"normalizedScore":45.7176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1658","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":23.911,"normalizedScore":23.911,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1659","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":87.6768,"normalizedScore":87.6768,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1660","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":68.8439,"normalizedScore":68.8439,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1661","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":81.9048,"normalizedScore":81.9048,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1662","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":92.6667,"normalizedScore":92.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1663","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":53.6735,"normalizedScore":53.6735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1664","modelSlug":"kimi-k2-thinking","modelName":"Kimi K2 Thinking","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66.3333,"normalizedScore":66.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1665","modelSlug":"kimi-k2-thinking","modelName":"Kimi K2 Thinking","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-20.4833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Kimi K2 Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1666","modelSlug":"kimi-k2-thinking","modelName":"Kimi K2 Thinking","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":2.5714,"normalizedScore":2.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 Thinking","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1667","modelSlug":"kimi-k2-thinking","modelName":"Kimi K2 Thinking","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":92.9825,"normalizedScore":92.9825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1668","modelSlug":"kimi-k2-thinking","modelName":"Kimi K2 Thinking","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":31.0606,"normalizedScore":31.0606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1669","modelSlug":"kimi-k2-thinking","modelName":"Kimi K2 Thinking","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":42.3611,"normalizedScore":42.3611,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1670","modelSlug":"kimi-k2-thinking","modelName":"Kimi K2 Thinking","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":22.3355,"normalizedScore":22.3355,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1671","modelSlug":"kimi-k2-thinking","modelName":"Kimi K2 Thinking","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":83.8384,"normalizedScore":83.8384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1672","modelSlug":"kimi-k2-thinking","modelName":"Kimi K2 Thinking","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":85.291,"normalizedScore":85.291,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1673","modelSlug":"kimi-k2-thinking","modelName":"Kimi K2 Thinking","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":94.6667,"normalizedScore":94.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1674","modelSlug":"kimi-k2-thinking","modelName":"Kimi K2 Thinking","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":68.0952,"normalizedScore":68.0952,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1675","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66,"normalizedScore":66,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3 Max Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1676","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-34.4167,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Qwen3 Max Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1677","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.7143,"normalizedScore":1.7143,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3 Max Thinking","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1678","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":83.6257,"normalizedScore":83.6257,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3 Max Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1679","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":24.2424,"normalizedScore":24.2424,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3 Max Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1680","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":43.0556,"normalizedScore":43.0556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3 Max Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1681","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":26.1816,"normalizedScore":26.1816,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3 Max Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1682","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":86.0606,"normalizedScore":86.0606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3 Max Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1683","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":70.7483,"normalizedScore":70.7483,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3 Max Thinking","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1684","modelSlug":"minimax-m2-1","modelName":"MiniMax M2.1","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":59,"normalizedScore":59,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1685","modelSlug":"minimax-m2-1","modelName":"MiniMax M2.1","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-32.8333,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1686","modelSlug":"minimax-m2-1","modelName":"MiniMax M2.1","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.2857,"normalizedScore":0.2857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.1","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1687","modelSlug":"minimax-m2-1","modelName":"MiniMax M2.1","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":85.3801,"normalizedScore":85.3801,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1688","modelSlug":"minimax-m2-1","modelName":"MiniMax M2.1","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":28.7879,"normalizedScore":28.7879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1689","modelSlug":"minimax-m2-1","modelName":"MiniMax M2.1","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.7407,"normalizedScore":40.7407,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1690","modelSlug":"minimax-m2-1","modelName":"MiniMax M2.1","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":22.1965,"normalizedScore":22.1965,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1691","modelSlug":"minimax-m2-1","modelName":"MiniMax M2.1","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":83.0303,"normalizedScore":83.0303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1692","modelSlug":"minimax-m2-1","modelName":"MiniMax M2.1","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":80.9524,"normalizedScore":80.9524,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1693","modelSlug":"minimax-m2-1","modelName":"MiniMax M2.1","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":82.6667,"normalizedScore":82.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1694","modelSlug":"minimax-m2-1","modelName":"MiniMax M2.1","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":69.8639,"normalizedScore":69.8639,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2.1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1695","modelSlug":"qwen3-5-omni-plus","modelName":"Qwen3.5 Omni Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":52.6667,"normalizedScore":52.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 Omni Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1696","modelSlug":"qwen3-5-omni-plus","modelName":"Qwen3.5 Omni Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-12.3,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Qwen3.5 Omni Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1697","modelSlug":"qwen3-5-omni-plus","modelName":"Qwen3.5 Omni Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.5633,"normalizedScore":0.5633,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 Omni Plus","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1698","modelSlug":"qwen3-5-omni-plus","modelName":"Qwen3.5 Omni Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":88.3041,"normalizedScore":88.3041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 Omni Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1699","modelSlug":"qwen3-5-omni-plus","modelName":"Qwen3.5 Omni Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":21.2121,"normalizedScore":21.2121,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 Omni Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1700","modelSlug":"qwen3-5-omni-plus","modelName":"Qwen3.5 Omni Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.5093,"normalizedScore":40.5093,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 Omni Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1701","modelSlug":"qwen3-5-omni-plus","modelName":"Qwen3.5 Omni Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":13.9018,"normalizedScore":13.9018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 Omni Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1702","modelSlug":"qwen3-5-omni-plus","modelName":"Qwen3.5 Omni Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":82.6263,"normalizedScore":82.6263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 Omni Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1703","modelSlug":"qwen3-5-omni-plus","modelName":"Qwen3.5 Omni Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":70.5202,"normalizedScore":70.5202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 Omni Plus","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1704","modelSlug":"qwen3-5-omni-plus","modelName":"Qwen3.5 Omni Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":51.1565,"normalizedScore":51.1565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 Omni Plus","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1705","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":62.6667,"normalizedScore":62.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 35B A3B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1706","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-46.3833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Qwen3.5 35B A3B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1707","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.8571,"normalizedScore":0.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 35B A3B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1708","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":89.1813,"normalizedScore":89.1813,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 35B A3B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1709","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":26.5152,"normalizedScore":26.5152,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 35B A3B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1710","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":21.516,"normalizedScore":21.516,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 35B A3B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1711","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":37.7315,"normalizedScore":37.7315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 35B A3B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1712","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":19.7405,"normalizedScore":19.7405,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 35B A3B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1713","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":84.5455,"normalizedScore":84.5455,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 35B A3B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1714","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":72.659,"normalizedScore":72.659,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 35B A3B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1715","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":72.517,"normalizedScore":72.517,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.5 35B A3B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1716","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":18.0275,"normalizedScore":18.0275,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1717","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":64.6667,"normalizedScore":64.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1718","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":0.1833,"normalizedScore":0.1833,"unit":"index","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1719","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.2857,"normalizedScore":0.2857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1720","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":64.6199,"normalizedScore":64.6199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1721","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":13.8144,"normalizedScore":13.8144,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1722","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":31.0606,"normalizedScore":31.0606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1723","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":36.3296,"normalizedScore":36.3296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1724","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.0463,"normalizedScore":40.0463,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1725","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":9.5922,"normalizedScore":9.5922,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1726","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":77.6768,"normalizedScore":77.6768,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1727","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":61.7919,"normalizedScore":61.7919,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1728","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":65.5026,"normalizedScore":65.5026,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1729","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":74.3333,"normalizedScore":74.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1730","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":54.6939,"normalizedScore":54.6939,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 4 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1731","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":21.6785,"normalizedScore":21.6785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1732","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":54.3333,"normalizedScore":54.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1733","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-41.7333,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1734","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":1.1429,"normalizedScore":1.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1735","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":70.4678,"normalizedScore":70.4678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1736","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":10.5155,"normalizedScore":10.5155,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1737","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":25,"normalizedScore":25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1738","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":49.4382,"normalizedScore":49.4382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1739","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":38.4259,"normalizedScore":38.4259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1740","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":13.3457,"normalizedScore":13.3457,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1741","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":77.9798,"normalizedScore":77.9798,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1742","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":69.5238,"normalizedScore":69.5238,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1743","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":86,"normalizedScore":86,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1744","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":43.4014,"normalizedScore":43.4014,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GLM-4.6 (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1745","modelSlug":"minimax-m2","modelName":"MiniMax M2","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":61,"normalizedScore":61,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1746","modelSlug":"minimax-m2","modelName":"MiniMax M2","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-46.9333,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"MiniMax-M2","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1747","modelSlug":"minimax-m2","modelName":"MiniMax M2","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.8571,"normalizedScore":0.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1748","modelSlug":"minimax-m2","modelName":"MiniMax M2","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":86.8421,"normalizedScore":86.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1749","modelSlug":"minimax-m2","modelName":"MiniMax M2","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":25.7576,"normalizedScore":25.7576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1750","modelSlug":"minimax-m2","modelName":"MiniMax M2","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":36.1111,"normalizedScore":36.1111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1751","modelSlug":"minimax-m2","modelName":"MiniMax M2","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":12.4652,"normalizedScore":12.4652,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1752","modelSlug":"minimax-m2","modelName":"MiniMax M2","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":77.6768,"normalizedScore":77.6768,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1753","modelSlug":"minimax-m2","modelName":"MiniMax M2","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":82.6455,"normalizedScore":82.6455,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1754","modelSlug":"minimax-m2","modelName":"MiniMax M2","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":78.3333,"normalizedScore":78.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1755","modelSlug":"minimax-m2","modelName":"MiniMax M2","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":72.3129,"normalizedScore":72.3129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MiniMax-M2","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1756","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":64.6667,"normalizedScore":64.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1757","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-28.4,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1758","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":2.8571,"normalizedScore":2.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1759","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":65.7895,"normalizedScore":65.7895,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1760","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":18.9394,"normalizedScore":18.9394,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1761","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":44.213,"normalizedScore":44.213,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1762","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":16.9601,"normalizedScore":16.9601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1763","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":84.7475,"normalizedScore":84.7475,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1764","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":61.7919,"normalizedScore":61.7919,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1765","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":83.1746,"normalizedScore":83.1746,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1766","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":89.6667,"normalizedScore":89.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1767","modelSlug":"grok-4-fast","modelName":"Grok 4 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":50.5442,"normalizedScore":50.5442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 4 Fast (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1768","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":35.871,"normalizedScore":35.871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1769","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":60.6667,"normalizedScore":60.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1770","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":0.3167,"normalizedScore":0.3167,"unit":"index","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1771","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.8571,"normalizedScore":0.8571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1772","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":54.6784,"normalizedScore":54.6784,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1773","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":21.2121,"normalizedScore":21.2121,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1774","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.2778,"normalizedScore":40.2778,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1775","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":10.2832,"normalizedScore":10.2832,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1776","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":77.1717,"normalizedScore":77.1717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1777","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":47.3016,"normalizedScore":47.3016,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1778","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":56.3333,"normalizedScore":56.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1779","modelSlug":"claude-sonnet-3-7","modelName":"Claude Sonnet 3.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":48.2993,"normalizedScore":48.2993,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude 3.7 Sonnet (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1780","modelSlug":"ling-2-6-1t","modelName":"Ling 2.6 1T","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Ling-2.6-1T","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1781","modelSlug":"ling-2-6-1t","modelName":"Ling 2.6 1T","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":34.6667,"normalizedScore":34.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Ling-2.6-1T","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1782","modelSlug":"ling-2-6-1t","modelName":"Ling 2.6 1T","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-51,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Ling-2.6-1T","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1783","modelSlug":"ling-2-6-1t","modelName":"Ling 2.6 1T","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.2857,"normalizedScore":0.2857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Ling-2.6-1T","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1784","modelSlug":"ling-2-6-1t","modelName":"Ling 2.6 1T","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":89.7661,"normalizedScore":89.7661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Ling-2.6-1T","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1785","modelSlug":"ling-2-6-1t","modelName":"Ling 2.6 1T","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":31.0606,"normalizedScore":31.0606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Ling-2.6-1T","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1786","modelSlug":"ling-2-6-1t","modelName":"Ling 2.6 1T","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":37.037,"normalizedScore":37.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Ling-2.6-1T","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1787","modelSlug":"ling-2-6-1t","modelName":"Ling 2.6 1T","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":8.2484,"normalizedScore":8.2484,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Ling-2.6-1T","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1788","modelSlug":"ling-2-6-1t","modelName":"Ling 2.6 1T","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":75.1515,"normalizedScore":75.1515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Ling-2.6-1T","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1789","modelSlug":"ling-2-6-1t","modelName":"Ling 2.6 1T","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":56.8707,"normalizedScore":56.8707,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Ling-2.6-1T","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1790","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":8.2695,"normalizedScore":8.2695,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1791","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":66,"normalizedScore":66,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1792","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-14.3,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1793","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":2.5714,"normalizedScore":2.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1794","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":54.0936,"normalizedScore":54.0936,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1795","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":9.2784,"normalizedScore":9.2784,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1796","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":26.5152,"normalizedScore":26.5152,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1797","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":28.4644,"normalizedScore":28.4644,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1798","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":42.8241,"normalizedScore":42.8241,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1799","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":21.0843,"normalizedScore":21.0843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1800","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":84.4444,"normalizedScore":84.4444,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1801","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":74.9133,"normalizedScore":74.9133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1802","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":80.1058,"normalizedScore":80.1058,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1803","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":87.6667,"normalizedScore":87.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1804","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":48.7075,"normalizedScore":48.7075,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Pro","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1805","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":55,"normalizedScore":55,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1806","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-35.75,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1807","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.5714,"normalizedScore":0.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1808","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":55.5556,"normalizedScore":55.5556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1809","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":15.1515,"normalizedScore":15.1515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1810","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":46.5278,"normalizedScore":46.5278,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1811","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":17.5112,"normalizedScore":17.5112,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1812","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":78.3838,"normalizedScore":78.3838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1813","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":69.2486,"normalizedScore":69.2486,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1814","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":85.9259,"normalizedScore":85.9259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1815","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":90.6667,"normalizedScore":90.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1816","modelSlug":"o4-mini","modelName":"o4-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":68.7075,"normalizedScore":68.7075,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o4-mini (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1817","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":9.6625,"normalizedScore":9.6625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1818","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":60,"normalizedScore":60,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1819","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-42.0667,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1820","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":3.1429,"normalizedScore":3.1429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1821","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":67.8363,"normalizedScore":67.8363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1822","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":10.1031,"normalizedScore":10.1031,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau3-banking","sourceTitle":"τ³-Banking Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1823","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":28.7879,"normalizedScore":28.7879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1824","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":38.5768,"normalizedScore":38.5768,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1825","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":1.8437,"normalizedScore":1.8437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-apex-agents","sourceTitle":"APEX-Agents-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/apex-agents-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1826","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"itbench-aa","benchmarkName":"ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / IBM","benchmarkVersion":null,"score":1.1299,"normalizedScore":1.1299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-itbench","sourceTitle":"ITBench-AA Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/itbench-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1827","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":35.9954,"normalizedScore":35.9954,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1828","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":19.1844,"normalizedScore":19.1844,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1829","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":80,"normalizedScore":80,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1830","modelSlug":"nemotron-3-super","modelName":"Nemotron 3 Super","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":71.4966,"normalizedScore":71.4966,"unit":"percent","scoreDirection":"higher","modelConfiguration":"NVIDIA Nemotron 3 Super 120B A12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1831","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":64.3333,"normalizedScore":64.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1832","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-34.7167,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1833","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.2857,"normalizedScore":0.2857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1834","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":45.614,"normalizedScore":45.614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1835","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":16.6667,"normalizedScore":16.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1836","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.5093,"normalizedScore":40.5093,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1837","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":12.7433,"normalizedScore":12.7433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1838","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":79.2929,"normalizedScore":79.2929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1839","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":73.1214,"normalizedScore":73.1214,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1840","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":71.3228,"normalizedScore":71.3228,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1841","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":78.3333,"normalizedScore":78.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1842","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":52.3129,"normalizedScore":52.3129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 2.5 Flash Preview (Sep '25) (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1843","modelSlug":"kimi-k2-0905","modelName":"Kimi K2 0905","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":52.3333,"normalizedScore":52.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 0905","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1844","modelSlug":"kimi-k2-0905","modelName":"Kimi K2 0905","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-26.4667,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Kimi K2 0905","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1845","modelSlug":"kimi-k2-0905","modelName":"Kimi K2 0905","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 0905","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1846","modelSlug":"kimi-k2-0905","modelName":"Kimi K2 0905","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":73.3918,"normalizedScore":73.3918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 0905","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1847","modelSlug":"kimi-k2-0905","modelName":"Kimi K2 0905","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":23.4848,"normalizedScore":23.4848,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 0905","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1848","modelSlug":"kimi-k2-0905","modelName":"Kimi K2 0905","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":30.6713,"normalizedScore":30.6713,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 0905","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1849","modelSlug":"kimi-k2-0905","modelName":"Kimi K2 0905","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":6.3485,"normalizedScore":6.3485,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 0905","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1850","modelSlug":"kimi-k2-0905","modelName":"Kimi K2 0905","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":76.6667,"normalizedScore":76.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 0905","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1851","modelSlug":"kimi-k2-0905","modelName":"Kimi K2 0905","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":60.9524,"normalizedScore":60.9524,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 0905","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1852","modelSlug":"kimi-k2-0905","modelName":"Kimi K2 0905","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":57.3333,"normalizedScore":57.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 0905","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1853","modelSlug":"kimi-k2-0905","modelName":"Kimi K2 0905","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":41.7007,"normalizedScore":41.7007,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K2 0905","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1854","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":59.3333,"normalizedScore":59.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1855","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-10.55,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"o1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1856","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.2857,"normalizedScore":0.2857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o1","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1857","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":62.5731,"normalizedScore":62.5731,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1858","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":12.8788,"normalizedScore":12.8788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1859","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":35.7639,"normalizedScore":35.7639,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1860","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":7.7496,"normalizedScore":7.7496,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1861","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":74.7475,"normalizedScore":74.7475,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1862","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":67.9365,"normalizedScore":67.9365,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1863","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":70.3401,"normalizedScore":70.3401,"unit":"percent","scoreDirection":"higher","modelConfiguration":"o1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1864","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":50.3333,"normalizedScore":50.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 3 mini Reasoning (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1865","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-6.05,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Grok 3 mini Reasoning (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1866","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0.5714,"normalizedScore":0.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 3 mini Reasoning (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1867","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":90.3509,"normalizedScore":90.3509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 3 mini Reasoning (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1868","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":17.4242,"normalizedScore":17.4242,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 3 mini Reasoning (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1869","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.625,"normalizedScore":40.625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 3 mini Reasoning (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1870","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":11.0656,"normalizedScore":11.0656,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 3 mini Reasoning (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1871","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":79.0909,"normalizedScore":79.0909,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 3 mini Reasoning (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1872","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":69.6296,"normalizedScore":69.6296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 3 mini Reasoning (high)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1873","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":84.6667,"normalizedScore":84.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 3 mini Reasoning (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1874","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":45.8503,"normalizedScore":45.8503,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok 3 mini Reasoning (high)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1875","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":55.3333,"normalizedScore":55.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1876","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-51.8833,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Gemma 4 12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1877","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1878","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":36.2573,"normalizedScore":36.2573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1879","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":18.1818,"normalizedScore":18.1818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1880","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":38.1944,"normalizedScore":38.1944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1881","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":14.7822,"normalizedScore":14.7822,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1882","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":75.2525,"normalizedScore":75.2525,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1883","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":69.6532,"normalizedScore":69.6532,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1884","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":73.5374,"normalizedScore":73.5374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemma 4 12B (Reasoning)","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1885","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":2.7365,"normalizedScore":2.7365,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-gdpval-aa","sourceTitle":"GDPval-AA v2 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1886","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":48.3333,"normalizedScore":48.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-lcr-leaderboard","sourceTitle":"AA Long Context Reasoning Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/artificial-analysis-long-context-reasoning","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1887","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":-36,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-omniscience-leaderboard","sourceTitle":"AA-Omniscience Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/omniscience","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1888","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1889","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2","score":75.731,"normalizedScore":75.731,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-tau2-bench","sourceTitle":"τ²-Bench Telecom Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/tau2-bench","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1890","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":17.4242,"normalizedScore":17.4242,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-terminalbench-hard","sourceTitle":"Terminal-Bench Hard Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/terminalbench-hard","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1891","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":36.2269,"normalizedScore":36.2269,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1892","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":7.4606,"normalizedScore":7.4606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1893","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":72.7273,"normalizedScore":72.7273,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1894","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":null,"score":65.7143,"normalizedScore":65.7143,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1895","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":null,"score":43.3333,"normalizedScore":43.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1896","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":41.3605,"normalizedScore":41.3605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Grok Code Fast 1","methodologyVersion":"1.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-15","publishedAt":"2026-07-15","sourceId":"aa-models-leaderboard","sourceTitle":"Artificial Analysis Models Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/leaderboards/models","sourceDate":"2026-07-15","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-15","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1897","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":57.9,"normalizedScore":57.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-06","publishedAt":"2026-07-06","sourceId":"tencent-hy3-hf","sourceTitle":"Tencent Hy3 model card on Hugging Face","sourcePublisher":"Tencent Hy Team","sourceUrl":"https://huggingface.co/tencent/Hy3","sourceDate":"2026-07-06","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1898","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified","score":78,"normalizedScore":78,"unit":"percent","scoreDirection":"higher","modelConfiguration":"tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-06","publishedAt":"2026-07-06","sourceId":"tencent-hy3-hf","sourceTitle":"Tencent Hy3 model card on Hugging Face","sourcePublisher":"Tencent Hy Team","sourceUrl":"https://huggingface.co/tencent/Hy3","sourceDate":"2026-07-06","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1899","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":25.6,"normalizedScore":25.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-06","publishedAt":"2026-07-06","sourceId":"tencent-hy3-hf","sourceTitle":"Tencent Hy3 model card on Hugging Face","sourcePublisher":"Tencent Hy Team","sourceUrl":"https://huggingface.co/tencent/Hy3","sourceDate":"2026-07-06","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1900","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":90.4,"normalizedScore":90.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-06","publishedAt":"2026-07-06","sourceId":"tencent-hy3-hf","sourceTitle":"Tencent Hy3 model card on Hugging Face","sourcePublisher":"Tencent Hy Team","sourceUrl":"https://huggingface.co/tencent/Hy3","sourceDate":"2026-07-06","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1901","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":53.2,"normalizedScore":53.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"tencent/Hy3 instruct weights; configuration as recorded on the Hugging Face evaluation results entry.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-06","publishedAt":"2026-07-06","sourceId":"tencent-hy3-hf","sourceTitle":"Tencent Hy3 model card on Hugging Face","sourcePublisher":"Tencent Hy Team","sourceUrl":"https://huggingface.co/tencent/Hy3","sourceDate":"2026-07-06","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1902","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":80,"normalizedScore":80,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Terminal-Bench 2.0).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1903","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":80.8,"normalizedScore":80.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (OSWorld-Verified).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1904","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":88.1,"normalizedScore":88.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (MCP Atlas).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1905","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":75.6,"normalizedScore":75.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Toolathlon).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1906","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"webarena","benchmarkName":"WebArena","benchmarkCategory":"agents","benchmarkOrganisation":"WebArena authors","benchmarkVersion":null,"score":69,"normalizedScore":69,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (WebArena-Verified).","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1907","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"v2","score":57.2,"normalizedScore":57.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Finance Agent v2).","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1908","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"cybench","benchmarkName":"Cybench","benchmarkCategory":"agents","benchmarkOrganisation":"Stanford / Cybench authors","benchmarkVersion":null,"score":92.9,"normalizedScore":92.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Cybench).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1909","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026-05","score":0.8,"normalizedScore":0.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (ExploitGym).","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1910","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":61.5,"normalizedScore":61.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (SWE-bench Pro).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1911","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":62.1,"normalizedScore":62.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (Humanity's Last Exam (with tools as published)).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1912","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":88.4,"normalizedScore":88.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (CharXiv Reasoning).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1913","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2","score":54.1,"normalizedScore":54.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; provider evaluation configuration as published in the Meta Muse Spark 1.1 evaluation report (MRCR 1M).","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"meta-muse-spark-1-1-eval","sourceTitle":"Meta AI Muse Spark 1.1 evaluation report","sourcePublisher":"Meta","sourceUrl":"https://ai.meta.com/static-resource/muse-spark-1-1-evaluation-report","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1914","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":58.2,"normalizedScore":58.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-model-muse-spark-1-1","sourceTitle":"Muse Spark 1.1 (xhigh) analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/muse-spark-1-1","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1915","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":77.9,"normalizedScore":77.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-model-muse-spark-1-1","sourceTitle":"Muse Spark 1.1 (xhigh) analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/muse-spark-1-1","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1916","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":89.8,"normalizedScore":89.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-model-muse-spark-1-1","sourceTitle":"Muse Spark 1.1 (xhigh) analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/muse-spark-1-1","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1917","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":45.1,"normalizedScore":45.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-hle-leaderboard","sourceTitle":"Humanity's Last Exam leaderboard (AA)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/humanitys-last-exam","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1918","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":63.3,"normalizedScore":63.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-model-muse-spark-1-1","sourceTitle":"Muse Spark 1.1 (xhigh) analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/muse-spark-1-1","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1919","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":15.1,"normalizedScore":15.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-critpt-leaderboard","sourceTitle":"CritPt Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/critpt","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-15","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1920","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":40.6,"normalizedScore":40.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-model-muse-spark-1-1","sourceTitle":"Muse Spark 1.1 (xhigh) analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/muse-spark-1-1","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1921","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":43.7,"normalizedScore":43.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-model-muse-spark-1-1","sourceTitle":"Muse Spark 1.1 (xhigh) analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/muse-spark-1-1","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1922","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":42.8,"normalizedScore":42.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-model-muse-spark-1-1","sourceTitle":"Muse Spark 1.1 (xhigh) analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/muse-spark-1-1","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1923","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":8.3,"normalizedScore":8.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-model-muse-spark-1-1","sourceTitle":"Muse Spark 1.1 (xhigh) analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/muse-spark-1-1","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1924","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":25.2,"normalizedScore":25.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1 (xhigh); Artificial Analysis independent evaluation configuration as published.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-model-muse-spark-1-1","sourceTitle":"Muse Spark 1.1 (xhigh) analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/muse-spark-1-1","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1925","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":76.18,"normalizedScore":76.18,"unit":"percent","scoreDirection":"higher","modelConfiguration":"mini-SWE-agent; reasoning xhigh; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-09","publishedAt":"2026-07-09","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-07-09","sourceCheckedAt":"2026-07-16","checkedAt":"2026-08-16","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-1926","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":73.9,"normalizedScore":73.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Terminus 2; reasoning high; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.4.1","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-05-01","publishedAt":"2026-05-01","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-05-01","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1927","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":65.8,"normalizedScore":65.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini CLI; reasoning high; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.4.1","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-05-01","publishedAt":"2026-05-01","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-05-01","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1928","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":58.7,"normalizedScore":58.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Code; reasoning max; Terminal-Bench 2.1 verified submission.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-05-01","publishedAt":"2026-05-01","sourceId":"terminal-bench-21","sourceTitle":"Terminal-Bench 2.1 leaderboard","sourcePublisher":"Terminal-Bench","sourceUrl":"https://www.tbench.ai/leaderboard/terminal-bench/2.1","sourceDate":"2026-05-01","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"official-leaderboard"},{"resultId":"evidence-2026-07-1929","modelSlug":"gpt-5","modelName":"GPT-5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"math-500","benchmarkName":"MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"OpenAI / Hendrycks","benchmarkVersion":null,"score":99.4,"normalizedScore":99.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis MATH-500 evaluation; reasoning setting high as stated on the MATH-500 leaderboard summary.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-math-500","sourceTitle":"MATH-500 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/math-500","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1930","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"math-500","benchmarkName":"MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"OpenAI / Hendrycks","benchmarkVersion":null,"score":99.2,"normalizedScore":99.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis MATH-500 evaluation; reasoning setting as published as stated on the MATH-500 leaderboard summary.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-math-500","sourceTitle":"MATH-500 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/math-500","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1931","modelSlug":"grok-3-mini","modelName":"Grok 3 mini","providerId":"xai","providerName":"xAI","benchmarkSlug":"math-500","benchmarkName":"MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"OpenAI / Hendrycks","benchmarkVersion":null,"score":99.2,"normalizedScore":99.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis MATH-500 evaluation; reasoning setting high as stated on the MATH-500 leaderboard summary.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-math-500","sourceTitle":"MATH-500 Leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/math-500","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1932","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":53.3,"normalizedScore":53.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis HLE evaluation; configuration Adaptive Reasoning, Max Effort, Opus 4.8 Fallback.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-hle-leaderboard","sourceTitle":"Humanity's Last Exam leaderboard (AA)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/humanitys-last-exam","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1933","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":47.2,"normalizedScore":47.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis HLE evaluation; configuration max.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-hle-leaderboard","sourceTitle":"Humanity's Last Exam leaderboard (AA)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/humanitys-last-exam","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1934","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":45.7,"normalizedScore":45.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis HLE evaluation; configuration Adaptive Reasoning, Max Effort.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-hle-leaderboard","sourceTitle":"Humanity's Last Exam leaderboard (AA)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/humanitys-last-exam","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1935","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-briefcase","benchmarkName":"AA-Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.3,"normalizedScore":86.3,"unit":"elo-proxy","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.1; AA-Briefcase Elo 863 reported via Artificial Analysis; stored as normalized display proxy (Elo/10) pending fixed transform.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"aa-briefcase","sourceTitle":"AA-Briefcase evaluation leaderboard","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/aa-briefcase","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-16","checkedAt":"2026-07-16","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1936","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":88.3,"normalizedScore":88.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"KimiCode harness; reasoning max; Terminal-Bench 2.1 as published in the Kimi K3 launch blog.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1937","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":91.2,"normalizedScore":91.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BrowseComp with context compaction triggered at 300K tokens; max reasoning; as stated in the Kimi K3 launch blog.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1938","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":84.2,"normalizedScore":84.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MCP Atlas 500-task public subset; 100-turn limit; Gemini 3.1 Pro judge; max reasoning.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1939","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":73.2,"normalizedScore":73.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Toolathlon-Verified; max reasoning; as published in the Kimi K3 launch blog.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1940","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":null,"score":37.6,"normalizedScore":37.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"APEX-Agents; max reasoning; provider-published Kimi K3 evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1941","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":56,"normalizedScore":56,"unit":"percent","scoreDirection":"higher","modelConfiguration":"HLE-Full with tools; max reasoning; provider-published Kimi K3 evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1942","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":93.5,"normalizedScore":93.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPQA Diamond; max reasoning; provider-published Kimi K3 evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1943","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":81.6,"normalizedScore":81.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"MMMU-Pro official protocol; original input order; images prepended to text; max reasoning; average of three runs.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1944","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":91.3,"normalizedScore":91.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"CharXiv Reasoning with Python; max reasoning; average of three runs.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1945","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":59.2,"normalizedScore":59.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K3; Artificial Analysis GDPval-AA v2 normalized score.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"aa-model-kimi-k3","sourceTitle":"Kimi K3 Intelligence, Performance & Price Analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1946","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":58.7,"normalizedScore":58.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K3; Artificial Analysis SciCode evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"aa-model-kimi-k3","sourceTitle":"Kimi K3 Intelligence, Performance & Price Analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1947","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":74.7,"normalizedScore":74.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K3; Artificial Analysis Long Context Reasoning.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"aa-model-kimi-k3","sourceTitle":"Kimi K3 Intelligence, Performance & Price Analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1948","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":23.4,"normalizedScore":23.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K3; Artificial Analysis CritPt evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"aa-model-kimi-k3","sourceTitle":"Kimi K3 Intelligence, Performance & Price Analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1949","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":18.4,"normalizedScore":18.4,"unit":"index","scoreDirection":"higher","modelConfiguration":"Kimi K3; Artificial Analysis Omniscience Index.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"aa-model-kimi-k3","sourceTitle":"Kimi K3 Intelligence, Performance & Price Analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1950","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":33.4,"normalizedScore":33.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K3; Artificial Analysis τ³-Banking evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"aa-model-kimi-k3","sourceTitle":"Kimi K3 Intelligence, Performance & Price Analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1951","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":52.7,"normalizedScore":52.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K3; Artificial Analysis AutomationBench-AA evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"aa-model-kimi-k3","sourceTitle":"Kimi K3 Intelligence, Performance & Price Analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1952","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"enterpriseops-gym-aa","benchmarkName":"EnterpriseOps-Gym-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / ServiceNow","benchmarkVersion":null,"score":45.3,"normalizedScore":45.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K3; Artificial Analysis EnterpriseOps-Gym-AA evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"aa-model-kimi-k3","sourceTitle":"Kimi K3 Intelligence, Performance & Price Analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1953","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":26.7,"normalizedScore":26.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K3; Artificial Analysis Harvey LAB-AA evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"aa-model-kimi-k3","sourceTitle":"Kimi K3 Intelligence, Performance & Price Analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1954","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":null,"score":85,"normalizedScore":85,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Kimi K3; Artificial Analysis Terminal-Bench Hard / Terminal-Bench v2.1 independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"aa-model-kimi-k3","sourceTitle":"Kimi K3 Intelligence, Performance & Price Analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1955","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":null,"score":63.3,"normalizedScore":63.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Code harness; OfficeQA Pro with PDF corpus rendered as images; max reasoning.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1956","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"v1.1","score":67.3,"normalizedScore":67.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"mini-SWE-agent common harness; DeepSWE v1.1; max reasoning. Provider also reports 67.5 with KimiCode.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1957","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"program-bench","benchmarkName":"ProgramBench","benchmarkCategory":"coding","benchmarkOrganisation":"ProgramBench","benchmarkVersion":"public","score":77.8,"normalizedScore":77.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"KimiCode harness; ProgramBench raw hidden-test pass rate; max reasoning.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1958","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-marathon","benchmarkName":"SWE Marathon","benchmarkCategory":"coding","benchmarkOrganisation":"Abundant AI","benchmarkVersion":"v1.1","score":42,"normalizedScore":42,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Code harness; SWE Marathon v1.1 H20-calibrated branch; max reasoning.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1959","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"frontierswe","benchmarkName":"FrontierSWE","benchmarkCategory":"coding","benchmarkOrganisation":"FrontierSWE","benchmarkVersion":"2026-07","score":81.2,"normalizedScore":81.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"KimiCode harness; FrontierSWE dominance recomputed with official script as of 2026-07-16.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1960","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"posttrain-bench","benchmarkName":"PostTrainBench","benchmarkCategory":"coding","benchmarkOrganisation":"PostTrainBench","benchmarkVersion":"public","score":36.6,"normalizedScore":36.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude Code harness; official Harbor implementation; max reasoning; average of three H20 runs.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1961","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"kimi-code-bench-v2","benchmarkName":"Kimi Code Bench v2","benchmarkCategory":"coding","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2.0","score":72.9,"normalizedScore":72.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"KimiCode and Claude Code harnesses at max effort; internal Kimi Code Bench 2.0.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-16","publishedAt":"2026-07-16","sourceId":"moonshot-kimi-k3-blog","sourceTitle":"Kimi K3: Open Frontier Intelligence","sourcePublisher":"Moonshot AI","sourceUrl":"https://www.kimi.com/blog/kimi-k3","sourceDate":"2026-07-16","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1962","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-briefcase","benchmarkName":"AA-Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":152.7,"normalizedScore":7.635,"unit":"elo-proxy","scoreDirection":"higher","modelConfiguration":"Kimi K3; Artificial Analysis AA-Briefcase Elo.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"aa-model-kimi-k3","sourceTitle":"Kimi K3 Intelligence, Performance & Price Analysis","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/kimi-k3","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1963","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":83.4,"normalizedScore":83.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-browsecomp-2026-07-20","sourceTitle":"BrowseComp Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/browsecomp","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1964","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":83.2,"normalizedScore":83.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-browsecomp-2026-07-20","sourceTitle":"BrowseComp Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/browsecomp","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1965","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":79.3,"normalizedScore":79.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-browsecomp-2026-07-20","sourceTitle":"BrowseComp Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/browsecomp","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1966","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":65.8,"normalizedScore":65.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-browsecomp-2026-07-20","sourceTitle":"BrowseComp Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/browsecomp","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1967","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":63.8,"normalizedScore":63.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-browsecomp-2026-07-20","sourceTitle":"BrowseComp Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/browsecomp","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1968","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":62,"normalizedScore":62,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-browsecomp-2026-07-20","sourceTitle":"BrowseComp Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/browsecomp","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1969","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":61,"normalizedScore":61,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-browsecomp-2026-07-20","sourceTitle":"BrowseComp Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/browsecomp","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1970","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":61,"normalizedScore":61,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-browsecomp-2026-07-20","sourceTitle":"BrowseComp Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/browsecomp","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1971","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":52,"normalizedScore":52,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-browsecomp-2026-07-20","sourceTitle":"BrowseComp Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/browsecomp","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1972","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":44.4,"normalizedScore":44.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public browsecomp leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-browsecomp-2026-07-20","sourceTitle":"BrowseComp Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/browsecomp","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1973","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":87.8,"normalizedScore":87.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-mmlu-pro-2026-07-20","sourceTitle":"MMLU-Pro Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/mmluPro","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1974","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":86.8,"normalizedScore":86.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-mmlu-pro-2026-07-20","sourceTitle":"MMLU-Pro Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/mmluPro","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1975","modelSlug":"qwen3-5-122b","modelName":"Qwen3.5 122B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":86.7,"normalizedScore":86.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-mmlu-pro-2026-07-20","sourceTitle":"MMLU-Pro Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/mmluPro","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1976","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":86.2,"normalizedScore":86.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-mmlu-pro-2026-07-20","sourceTitle":"MMLU-Pro Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/mmluPro","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1977","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":86.1,"normalizedScore":86.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-mmlu-pro-2026-07-20","sourceTitle":"MMLU-Pro Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/mmluPro","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1978","modelSlug":"qwen3-5-35b","modelName":"Qwen3.5 35B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":85.3,"normalizedScore":85.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-mmlu-pro-2026-07-20","sourceTitle":"MMLU-Pro Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/mmluPro","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1979","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":84.9,"normalizedScore":84.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-mmlu-pro-2026-07-20","sourceTitle":"MMLU-Pro Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/mmluPro","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"evidence-2026-07-1980","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":84.3,"normalizedScore":84.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-mmlu-pro-2026-07-20","sourceTitle":"MMLU-Pro Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/mmluPro","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1981","modelSlug":"gemma-4-26b","modelName":"Gemma 4 26B","providerId":"google","providerName":"Google","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":82.6,"normalizedScore":82.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-mmlu-pro-2026-07-20","sourceTitle":"MMLU-Pro Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/mmluPro","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1982","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"mmlu-pro","benchmarkName":"MMLU-Pro","benchmarkCategory":"research","benchmarkOrganisation":"TIGER-Lab","benchmarkVersion":null,"score":77.2,"normalizedScore":77.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public mmlu-pro leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-mmlu-pro-2026-07-20","sourceTitle":"MMLU-Pro Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/mmluPro","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1983","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.0/2.1","score":69.4,"normalizedScore":69.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public terminal-bench leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-terminal-bench-2026-07-20","sourceTitle":"Terminal-Bench 2.0 Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/terminalBench","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1984","modelSlug":"qwen3-6-max","modelName":"Qwen3.6 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.0/2.1","score":65.4,"normalizedScore":65.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public terminal-bench leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-terminal-bench-2026-07-20","sourceTitle":"Terminal-Bench 2.0 Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/terminalBench","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1985","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.0/2.1","score":54.4,"normalizedScore":54.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public terminal-bench leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-terminal-bench-2026-07-20","sourceTitle":"Terminal-Bench 2.0 Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/terminalBench","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1986","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":35.7,"normalizedScore":35.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public gdpval-aa leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-gdpval-aa-normalized-2026-07-20","sourceTitle":"GDPval-AA Normalized Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/gdpvalAaNormalized","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1987","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":21.8,"normalizedScore":21.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact model variant as listed on the BenchLM public gdpval-aa leaderboard page (verified 2026-07-21).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-20","publishedAt":"2026-07-20","sourceId":"benchlm-gdpval-aa-normalized-2026-07-20","sourceTitle":"GDPval-AA Normalized Leaderboard & Scores — July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/benchmarks/gdpvalAaNormalized","sourceDate":"2026-07-20","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1988","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":58.7,"normalizedScore":58.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). Internal Antigravity harness; self-computed per Google methodology.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1989","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"v1.1","score":49,"normalizedScore":49,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). High reasoning; public DeepSWE v1.1 leaderboard as cited by Google.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-blog","sourceTitle":"Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","sourcePublisher":"Google","sourceUrl":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1990","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":78,"normalizedScore":78,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). Terminus-2 harness; Terminal-Bench 2.1.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1991","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mle-bench","benchmarkName":"MLE-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"OpenAI","benchmarkVersion":"Partial-30","score":63.9,"normalizedScore":63.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). MLE-Bench Partial-30 Average Position Score; k=2 independent runs; bash terminal harness.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1992","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":83,"normalizedScore":83,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). OSWorld Verified; averaged over 5 runs; max 100 steps; 1080p.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1993","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":46.05,"normalizedScore":46.05,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). GDPval-AA v2 Elo as published by Google (sourced from AA leaderboard in methodology).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1994","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":85.2,"normalizedScore":85.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). CharXiv Reasoning without tools; self-computed.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1995","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":89.4,"normalizedScore":89.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). CharXiv Reasoning with search and code-execution tools.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1996","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2","score":91.8,"normalizedScore":91.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). GDM-MRCR v2 cumulative score at 128k context.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1997","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2","score":54,"normalizedScore":54,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.6-flash via Gemini API; default sampling; single-attempt / pass@1 unless noted in Google eval methodology (July 2026). GDM-MRCR v2 pointwise score at 1M context.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-1998","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":77.5,"normalizedScore":77.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis Terminal-Bench v2.1 independent evaluation (as listed on AA / BenchLM).","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-6-flash","sourceTitle":"Gemini 3.6 Flash (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-6-flash","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-1999","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.8,"normalizedScore":92.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis GPQA Diamond independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-6-flash","sourceTitle":"Gemini 3.6 Flash (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-6-flash","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2000","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":38.3,"normalizedScore":38.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis Humanity's Last Exam independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-6-flash","sourceTitle":"Gemini 3.6 Flash (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-6-flash","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2001","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":83.2,"normalizedScore":83.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis MMMU-Pro independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-6-flash","sourceTitle":"Gemini 3.6 Flash (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-6-flash","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2002","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":52.7,"normalizedScore":52.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis SciCode independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-6-flash","sourceTitle":"Gemini 3.6 Flash (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-6-flash","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2003","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":69.7,"normalizedScore":69.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis Long Context Reasoning independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-6-flash","sourceTitle":"Gemini 3.6 Flash (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-6-flash","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2004","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":10.6,"normalizedScore":10.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis CritPt independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-6-flash","sourceTitle":"Gemini 3.6 Flash (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-6-flash","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2005","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":24.5,"normalizedScore":24.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis τ³-Banking independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-6-flash","sourceTitle":"Gemini 3.6 Flash (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-6-flash","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2006","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":23.5,"normalizedScore":23.5,"unit":"index","scoreDirection":"higher","modelConfiguration":"Artificial Analysis Omniscience Index independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-6-flash","sourceTitle":"Gemini 3.6 Flash (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-6-flash","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2007","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-briefcase","benchmarkName":"AA-Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":96.136,"normalizedScore":96.136,"unit":"elo-proxy","scoreDirection":"higher","modelConfiguration":"Artificial Analysis AA-Briefcase Elo (mid).","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-6-flash","sourceTitle":"Gemini 3.6 Flash (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-6-flash","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2008","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":46.1,"normalizedScore":46.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis GDPval-AA v2 normalized score as published on AA model page.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-6-flash","sourceTitle":"Gemini 3.6 Flash (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-6-flash","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2009","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"v1.1","score":37,"normalizedScore":37,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table for Gemini 3.5 Flash comparison.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2010","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mle-bench","benchmarkName":"MLE-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"OpenAI","benchmarkVersion":"Partial-30","score":49.7,"normalizedScore":49.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table for Gemini 3.5 Flash comparison.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2011","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":42.45,"normalizedScore":42.45,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GDPval-AA v2 Elo 1349 as published in Gemini 3.6 evaluation table for 3.5 Flash.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2012","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"mle-bench","benchmarkName":"MLE-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"OpenAI","benchmarkVersion":"Partial-30","score":42.6,"normalizedScore":42.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table for Gemini 3.1 Pro comparison.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2013","modelSlug":"gemini-3-1-pro-preview","modelName":"Gemini 3.1 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":73.8,"normalizedScore":73.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Terminus-2; as published in Gemini 3.6 Flash evaluation table for Gemini 3.1 Pro.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2014","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":62.7,"normalizedScore":62.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table (provider-reported peer numbers).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2015","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"v1.1","score":67,"normalizedScore":67,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2016","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":84.7,"normalizedScore":84.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Terminus-2; as published in Gemini 3.6 Flash evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2017","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mle-bench","benchmarkName":"MLE-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"OpenAI","benchmarkVersion":"Partial-30","score":47.6,"normalizedScore":47.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2018","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":72.6,"normalizedScore":72.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2019","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":54.2,"normalizedScore":54.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GDPval-AA v2 Elo 1584 as published in Gemini 3.6 evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2020","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":82.7,"normalizedScore":82.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"CharXiv Reasoning no-tools; as published in Gemini 3.6 evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2021","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":64.7,"normalizedScore":64.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table (provider-reported peer numbers).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2022","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"v1.1","score":54,"normalizedScore":54,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2023","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":83.3,"normalizedScore":83.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Terminus-2; as published in Gemini 3.6 Flash evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2024","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"mle-bench","benchmarkName":"MLE-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"OpenAI","benchmarkVersion":"Partial-30","score":43.2,"normalizedScore":43.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2025","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":51.75,"normalizedScore":51.75,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GDPval-AA v2 Elo 1535 as published in Gemini 3.6 evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2026","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":81.6,"normalizedScore":81.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"CharXiv Reasoning no-tools; as published in Gemini 3.6 evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2027","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":63.2,"normalizedScore":63.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table (provider-reported peer numbers).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2028","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"v1.1","score":54,"normalizedScore":54,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2029","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":80.4,"normalizedScore":80.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Terminus-2; as published in Gemini 3.6 Flash evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2030","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mle-bench","benchmarkName":"MLE-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"OpenAI","benchmarkVersion":"Partial-30","score":66.9,"normalizedScore":66.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2031","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":81.2,"normalizedScore":81.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.6 Flash evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2032","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":55.35,"normalizedScore":55.35,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GDPval-AA v2 Elo 1607 as published in Gemini 3.6 evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2033","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"May 2026","score":77,"normalizedScore":77,"unit":"percent","scoreDirection":"higher","modelConfiguration":"CharXiv Reasoning no-tools; as published in Gemini 3.6 evaluation table.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-product","sourceTitle":"Gemini 3.6 Flash product page with evaluation table","sourcePublisher":"Google DeepMind","sourceUrl":"https://deepmind.google/models/gemini/flash/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2034","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":54,"normalizedScore":54,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. Terminal-Bench 2.1 as published in Google launch blog.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-blog","sourceTitle":"Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","sourcePublisher":"Google","sourceUrl":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2035","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2","score":72.2,"normalizedScore":72.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. GDM-MRCR v2 as published in Google launch blog.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-blog","sourceTitle":"Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","sourcePublisher":"Google","sourceUrl":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2036","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":32,"normalizedScore":32,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. GDPval-AA v2 Elo 1140 as published in Google launch blog.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-blog","sourceTitle":"Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","sourcePublisher":"Google","sourceUrl":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2037","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":54.2,"normalizedScore":54.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. SWE-Bench Pro as published in Google launch blog.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-blog","sourceTitle":"Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","sourcePublisher":"Google","sourceUrl":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2038","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":74,"normalizedScore":74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"gemini-3.5-flash-lite via Gemini API; configuration as stated in the July 2026 Google launch materials. OSWorld-Verified as published in Google launch blog.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-blog","sourceTitle":"Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","sourcePublisher":"Google","sourceUrl":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2039","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":83.8,"normalizedScore":83.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis GPQA Diamond independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-5-flash-lite","sourceTitle":"Gemini 3.5 Flash-Lite (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-lite","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2040","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":17.5,"normalizedScore":17.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis Humanity's Last Exam independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-5-flash-lite","sourceTitle":"Gemini 3.5 Flash-Lite (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-lite","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2041","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":79,"normalizedScore":79,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis MMMU-Pro independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-5-flash-lite","sourceTitle":"Gemini 3.5 Flash-Lite (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-lite","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2042","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":null,"score":40.9,"normalizedScore":40.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis SciCode independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-5-flash-lite","sourceTitle":"Gemini 3.5 Flash-Lite (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-lite","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2043","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":62,"normalizedScore":62,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis Long Context Reasoning independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-5-flash-lite","sourceTitle":"Gemini 3.5 Flash-Lite (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-lite","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2044","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":null,"score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis CritPt independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-5-flash-lite","sourceTitle":"Gemini 3.5 Flash-Lite (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-lite","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2045","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3","score":16.5,"normalizedScore":16.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis τ³-Banking independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-5-flash-lite","sourceTitle":"Gemini 3.5 Flash-Lite (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-lite","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2046","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":null,"score":6.9,"normalizedScore":6.9,"unit":"index","scoreDirection":"higher","modelConfiguration":"Artificial Analysis Omniscience Index independent evaluation.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-5-flash-lite","sourceTitle":"Gemini 3.5 Flash-Lite (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-lite","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2047","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-briefcase","benchmarkName":"AA-Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.4,"normalizedScore":63.4,"unit":"elo-proxy","scoreDirection":"higher","modelConfiguration":"Artificial Analysis AA-Briefcase Elo.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-5-flash-lite","sourceTitle":"Gemini 3.5 Flash-Lite (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-lite","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2048","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":32,"normalizedScore":32,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Artificial Analysis GDPval-AA v2 normalized score.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"aa-gemini-3-5-flash-lite","sourceTitle":"Gemini 3.5 Flash-Lite (Artificial Analysis)","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/models/gemini-3-5-flash-lite","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2049","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":49.6,"normalizedScore":49.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.5 Flash-Lite launch materials comparing against Gemini 3 Flash.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-blog","sourceTitle":"Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","sourcePublisher":"Google","sourceUrl":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2050","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":65.1,"normalizedScore":65.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.5 Flash-Lite launch materials comparing against Gemini 3 Flash.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-blog","sourceTitle":"Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","sourcePublisher":"Google","sourceUrl":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2051","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":31,"normalizedScore":31,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.5 Flash-Lite launch materials (prior Flash-Lite comparison).","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-blog","sourceTitle":"Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","sourcePublisher":"Google","sourceUrl":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2052","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2","score":60.1,"normalizedScore":60.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"As published in Gemini 3.5 Flash-Lite launch materials.","methodologyVersion":"1.4.1","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-blog","sourceTitle":"Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","sourcePublisher":"Google","sourceUrl":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2053","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":7.1,"normalizedScore":7.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GDPval-AA v2 Elo 642 as published for 3.1 Flash-Lite comparison.","methodologyVersion":"1.4.1","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"google-gemini-36-blog","sourceTitle":"Introducing Gemini 3.6 Flash, 3.5 Flash-Lite, and 3.5 Flash Cyber","sourcePublisher":"Google","sourceUrl":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-6-flash-3-5-flash-lite-3-5-flash-cyber/","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"provider-reported"},{"resultId":"benchlm-ref-claude-mythos-5-terminalbench2-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":88,"normalizedScore":93.0605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-osworldverified-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":85,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-browsecomp-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":88,"normalizedScore":91.2134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-exploitgym-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":17.5,"normalizedScore":50.7599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-sweverified-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-swepro-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":80.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-swemultimodal-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultimodal","benchmarkName":"SWE-bench Multimodal","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":54.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-charxiv-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":93.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-charxivnotools-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":88.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-gpqa-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":94.1,"normalizedScore":98.0026,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-hle-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":64.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-hlenotools-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":59,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-swemultilingual-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":92.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-usamo2026-2026-07-21","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":97.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-terminalbench2-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":74.6,"normalizedScore":69.2171,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-browsecomp-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":84.3,"normalizedScore":83.4728,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-deepsearchqa-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":93.1,"normalizedScore":94.0994,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-osworldverified-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":83.4,"normalizedScore":96.5217,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-financeagentv2-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":53.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gdpvalaa-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1594,"normalizedScore":91.8605,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-mcpatlas-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":82.2,"normalizedScore":89.9317,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-toolathlon-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":59.9,"normalizedScore":67.7618,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gertlabs-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":72.97,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaagenticindex-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.18,"normalizedScore":87.3069,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-tau2bench-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.4,"normalizedScore":95.2573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gdpvalaanormalized-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.7,"normalizedScore":87.6603,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-researchclawbench-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":21.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-osworld2-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":20.6,"normalizedScore":29.7659,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aabriefcaseelo-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1347,"normalizedScore":78.7453,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaautomationbench-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.5,"normalizedScore":84.5018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaenterpriseopsgym-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44,"normalizedScore":72.2656,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaharveylab-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.1,"normalizedScore":90.1961,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aatau3banking-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.6,"normalizedScore":69.4737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-terminalbenchhard-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":58.3,"normalizedScore":81.4181,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaterminalbench21-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.6,"normalizedScore":85.8333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-sweverified-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":88.6,"normalizedScore":90.4033,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-swepro-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":69.2,"normalizedScore":70.3209,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-swemultilingual-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":84.4,"normalizedScore":80.597,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-swemultimodal-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultimodal","benchmarkName":"SWE-bench Multimodal","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":38.4,"normalizedScore":38.4328,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aacodingindex-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.25,"normalizedScore":95.4637,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aascicode-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.5,"normalizedScore":88.7015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-frontiercode-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":46.5,"normalizedScore":76.0274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-lcr-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.7,"normalizedScore":89.432,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-critpt-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":20.9,"normalizedScore":64.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-officeqapro-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":66.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-screenspotpro-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":87.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-charxiv-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":89.9,"normalizedScore":91.1765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-charxivnotools-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":80.5,"normalizedScore":29.4118,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-designarenawebsite-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1270,"normalizedScore":80.8581,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gpqa-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gpqadiamond-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-hle-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":57.9,"normalizedScore":88.3803,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-hlenotools-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":49.8,"normalizedScore":82.8996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaomniscienceindex-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.4,"normalizedScore":89.9529,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-omniscienceaccuracy-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.6,"normalizedScore":74.5704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-omnisciencehallucinationrate-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.9,"normalizedScore":73.7033,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-include-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-include","benchmarkName":"INCLUDE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaifbench-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":62.2,"normalizedScore":68.6838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-usamo2026-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":96.7,"normalizedScore":92.4306,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-frontiermathv2tiers13-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":47.241,"normalizedScore":53.0798,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-frontiermathv2tier4-2026-07-21","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":31.25,"normalizedScore":37.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-terminalbench2-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":91.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-browsecomp-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":92.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-osworld2-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":62.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-cybergym-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":84.5,"normalizedScore":94.508,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-exploitgym-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":33.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-toolathlon-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":58,"normalizedScore":63.8604,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaagenticindex-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-tau2bench-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":85.1,"normalizedScore":85.8729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61.8,"normalizedScore":99.0385,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gdpvalaa-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1736,"normalizedScore":99.3658,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aabriefcaseelo-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1501,"normalizedScore":93.1648,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaitbench-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aatau3banking-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33,"normalizedScore":97.8947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaautomationbench-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.2,"normalizedScore":94.4649,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaharveylab-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.2,"normalizedScore":79.2717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-terminalbenchhard-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":65.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaterminalbench21-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-swepro-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":64.6,"normalizedScore":58.0214,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-deepswe-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":72.7,"normalizedScore":100,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiercode11extended-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode11extended","benchmarkName":"FrontierCode 1.1 Extended","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":60.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-vulcanbench-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vulcanbench","benchmarkName":"VulcanBench v3","benchmarkCategory":"coding","benchmarkOrganisation":"VulcanBench contributors","benchmarkVersion":"2026","score":87,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aacodingindex-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.39,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aascicode-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":93.086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-arcagi3-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":7.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-genebenchpro-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-genebenchpro","benchmarkName":"GeneBench-Pro","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":28.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-lcr-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.7,"normalizedScore":97.358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-critpt-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":32.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-mmmupro-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":83,"normalizedScore":64.5161,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-mmmupropython-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":84.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aammmupro-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":83.4,"normalizedScore":98.4429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gpqa-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":94.6,"normalizedScore":98.7159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gpqadiamond-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":94.6,"normalizedScore":98.7159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-healthbenchprofessional-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":60.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-healthbenchhard-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":33.1,"normalizedScore":65.3571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aahle-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":47.2,"normalizedScore":87.8486,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.7,"normalizedScore":85.4788,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.5,"normalizedScore":95.0172,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.8,"normalizedScore":9.8914,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaifbench-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.7,"normalizedScore":84.5688,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiermath-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":89,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":89,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiermathv2tier4-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":83,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-exploitbench-2026-07-21","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-exploitbench","benchmarkName":"ExploitBench v8-bench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Seunghyun Lee, David Brumley, Carnegie Mellon University","benchmarkVersion":"2026","score":73.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-terminalbench2-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":84.3,"normalizedScore":86.4769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-osworldverified-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":85,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-gdpvalaa-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1748,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaagenticindex-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.81,"normalizedScore":97.7852,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-tau2bench-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-gdpvalaanormalized-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aabriefcaseelo-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1574,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaautomationbench-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.6,"normalizedScore":84.8708,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaenterpriseopsgym-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaharveylab-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.6,"normalizedScore":97.1989,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aatau3banking-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":65.2632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-terminalbenchhard-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":62.9,"normalizedScore":92.665,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaterminalbench21-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.6,"normalizedScore":85.8333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-sweverified-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":95,"normalizedScore":99.3046,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-swepro-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":80,"normalizedScore":99.1979,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-frontiercode-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":53.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-vulcanbench-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vulcanbench","benchmarkName":"VulcanBench v3","benchmarkCategory":"coding","benchmarkOrganisation":"VulcanBench contributors","benchmarkVersion":"2026","score":87,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aacodingindex-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76.49,"normalizedScore":98.6998,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aascicode-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":60.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-lcr-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70,"normalizedScore":92.4703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-critpt-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":28.6,"normalizedScore":88.5449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-blueprintbench2-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"blueprint-bench-2","benchmarkName":"Blueprint-Bench 2","benchmarkCategory":"multimodal","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":38.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-officeqapro-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":57.9,"normalizedScore":63.2743,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-designarenawebsite-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1332,"normalizedScore":91.0891,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aagpqadiamond-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":92.6,"normalizedScore":97.9675,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aahle-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":53.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaomniscienceindex-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.2,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-omniscienceaccuracy-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-omnisciencehallucinationrate-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.9,"normalizedScore":50.7841,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaifbench-2026-07-21","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":63.5,"normalizedScore":70.6505,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-terminalbench2-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":80.4,"normalizedScore":79.5374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-browsecomp-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":84.7,"normalizedScore":84.3096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-hlewithtools-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":57.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-osworldverified-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":81.2,"normalizedScore":91.7391,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-gdpvalaa-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1607,"normalizedScore":92.5476,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaagenticindex-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.69,"normalizedScore":86.3949,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-gdpvalaanormalized-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.4,"normalizedScore":88.7821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aabriefcaseelo-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1388,"normalizedScore":82.5843,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaautomationbench-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.2,"normalizedScore":50.1845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaenterpriseopsgym-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.7,"normalizedScore":75,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaharveylab-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.1,"normalizedScore":87.395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aatau3banking-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.2,"normalizedScore":72.6316,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaterminalbench21-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.5,"normalizedScore":68.75,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-sweverified-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":85.2,"normalizedScore":85.6745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-swepro-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":63.2,"normalizedScore":54.2781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-swemultilingual-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.3,"normalizedScore":65.4229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-swemultimodal-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultimodal","benchmarkName":"SWE-bench Multimodal","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":28.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-frontiercode-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":42.7,"normalizedScore":63.0137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aacodingindex-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.55,"normalizedScore":91.5631,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aascicode-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.6,"normalizedScore":88.8702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-lcr-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70.7,"normalizedScore":93.395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-critpt-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":16.9,"normalizedScore":52.322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-charxiv-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":88.3,"normalizedScore":87.2549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-charxivnotools-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":77,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aammmupro-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":77.3,"normalizedScore":87.8893,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-designarenawebsite-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1314,"normalizedScore":88.1188,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-hle-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":57.4,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-hlenotools-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":43.2,"normalizedScore":70.632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aagpqadiamond-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":91.1,"normalizedScore":95.935,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaomniscienceindex-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.3,"normalizedScore":80.4553,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-omniscienceaccuracy-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.3,"normalizedScore":60.3093,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-omnisciencehallucinationrate-2026-07-21","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.3,"normalizedScore":72.0145,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-terminalbench2-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":69.7,"normalizedScore":60.4982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-qwenclawbench-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":64.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-qwenwebbench-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1568,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-claweval-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":65.2,"normalizedScore":83.3799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-bfclv4-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mcpatlas-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":76.4,"normalizedScore":80.0341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-vitabench-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":47.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-hlewithtools-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":53.5,"normalizedScore":80.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaagenticindex-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.59,"normalizedScore":56.4303,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-tau2bench-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.7,"normalizedScore":95.56,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gdpvalaanormalized-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.7,"normalizedScore":62.0192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gdpvalaa-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1273,"normalizedScore":74.8943,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gertlabs-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":64.27,"normalizedScore":81.6145,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-researchclawbench-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18.7,"normalizedScore":72.4138,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aabriefcaseelo-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":908,"normalizedScore":37.6404,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaautomationbench-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaenterpriseopsgym-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45,"normalizedScore":76.1719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaitbench-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.5,"normalizedScore":72.9249,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-terminalbenchhard-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":50.8,"normalizedScore":63.0807,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaterminalbench21-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.5,"normalizedScore":43.75,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaharveylab-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.4,"normalizedScore":68.6275,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-sweverified-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.4,"normalizedScore":78.9986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-swepro-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":60.6,"normalizedScore":47.3262,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-swemultilingual-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.3,"normalizedScore":65.4229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-nl2repo-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":47.2,"normalizedScore":92.1659,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-scicode-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":53.5,"normalizedScore":80.0604,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-livecodebench-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":91.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aacodingindex-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.97,"normalizedScore":83.5019,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aascicode-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":48.8,"normalizedScore":80.7757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mrcrv2-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":90.4,"normalizedScore":93.6255,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-critpt-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":13.4,"normalizedScore":41.4861,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-lcr-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69,"normalizedScore":91.1493,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-designarenawebsite-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1293,"normalizedScore":84.6535,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gpqa-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.4,"normalizedScore":95.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gpqadiamond-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.4,"normalizedScore":95.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-hle-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":41.4,"normalizedScore":59.331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmlupro-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":89.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmluredux-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":95,"normalizedScore":93.9714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-supergpqa-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":73.6,"normalizedScore":70.2199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmmlu-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90.3,"normalizedScore":92,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaomniscienceindex-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.1,"normalizedScore":79.5133,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-omniscienceaccuracy-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.1,"normalizedScore":46.2199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-omnisciencehallucinationrate-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.9,"normalizedScore":89.3848,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmluprox-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":87,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-nova63-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59,"normalizedScore":97.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-include-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-include","benchmarkName":"INCLUDE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":69.5652,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-maxife-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-maxife","benchmarkName":"MAXIFE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":89.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-polymath-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-polymath","benchmarkName":"PolyMath","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-ifeval-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.3,"normalizedScore":97.9314,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-ifbench-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":79.1,"normalizedScore":87.3391,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaifbench-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":80.5,"normalizedScore":96.3691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-hmmtfeb2026-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":97.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-imoanswerbench-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-apex-2026-07-21","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":44.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-terminalbench2-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":87.4,"normalizedScore":91.9929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-browsecomp-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":87.5,"normalizedScore":90.1674,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-osworld2-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":50.2,"normalizedScore":79.2642,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-cybergym-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":81.8,"normalizedScore":88.3295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-exploitgym-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":23.2,"normalizedScore":68.0851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-toolathlon-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":53.1,"normalizedScore":53.7988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaagenticindex-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.38,"normalizedScore":87.6791,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-tau2bench-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86.3,"normalizedScore":87.0838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.1,"normalizedScore":86.6987,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gdpvalaa-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1581,"normalizedScore":91.1734,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaharveylab-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.2,"normalizedScore":73.6695,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaitbench-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51,"normalizedScore":89.7233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aatau3banking-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.8,"normalizedScore":91.5789,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaautomationbench-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.6,"normalizedScore":73.8007,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-terminalbenchhard-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":57.6,"normalizedScore":79.7066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaterminalbench21-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-apexagentsaa-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":38.9,"normalizedScore":82.3276,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-swepro-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":63.4,"normalizedScore":54.8128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-deepswe-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":69.6,"normalizedScore":90.4025,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiercode11extended-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode11extended","benchmarkName":"FrontierCode 1.1 Extended","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":55.8,"normalizedScore":12.7273,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aacodingindex-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76.66,"normalizedScore":98.9454,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aascicode-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.9,"normalizedScore":89.3761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-arcagi3-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.8,"normalizedScore":7.8947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-lcr-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-critpt-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":30,"normalizedScore":92.8793,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-mmmupro-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":80.7,"normalizedScore":57.0968,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-mmmupropython-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":82,"normalizedScore":82.7815,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aammmupro-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.7,"normalizedScore":93.7716,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gpqa-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.9,"normalizedScore":96.2905,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gpqadiamond-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.9,"normalizedScore":96.2905,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-healthbenchprofessional-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":57.7,"normalizedScore":77.4194,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-healthbenchhard-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":32.7,"normalizedScore":63.9286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aahle-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":41.8,"normalizedScore":77.0916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-0.2,"normalizedScore":68.2889,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.9,"normalizedScore":73.3677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.2,"normalizedScore":14.234,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaifbench-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":71.2,"normalizedScore":82.2995,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiermath-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":84.9,"normalizedScore":90.9292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":84.9,"normalizedScore":95.3933,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiermathv2tier4-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":68.3,"normalizedScore":82.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-exploitbench-2026-07-21","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-exploitbench","benchmarkName":"ExploitBench v8-bench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Seunghyun Lee, David Brumley, Carnegie Mellon University","benchmarkVersion":"2026","score":52.9,"normalizedScore":48.8834,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-terminalbench2-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":88.3,"normalizedScore":93.5943,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-browsecomp-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":91.2,"normalizedScore":97.9079,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-deepsearchqa-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-toolathlonverified-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-toolathlonverified","benchmarkName":"Toolathlon-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":73.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mcpatlas-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":84.2,"normalizedScore":93.3447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-automationbench-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-automationbench","benchmarkName":"AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":30.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-jobbench-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":52.9,"normalizedScore":96.1014,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-apexagents-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-apexagents","benchmarkName":"APEX-Agents","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI / APEX-Agents benchmark authors","benchmarkVersion":"2026","score":37.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-spreadsheetbench2-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-spreadsheetbench2","benchmarkName":"SpreadsheetBench 2","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":34.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-deckbench-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deckbench","benchmarkName":"DECK-Bench (Internal)","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":73.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaagenticindex-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.07,"normalizedScore":92.6857,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gdpvalaanormalized-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59,"normalizedScore":94.5513,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gdpvalaa-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1679,"normalizedScore":96.3531,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aabriefcaseelo-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1543,"normalizedScore":97.0974,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaautomationbench-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaenterpriseopsgym-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.3,"normalizedScore":77.3437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaharveylab-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aatau3banking-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaterminalbench21-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-apexagentsaa-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":41.3,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaitbench-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.7,"normalizedScore":83.2016,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-deepswe-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":67.5,"normalizedScore":83.9009,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-frontierswe-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"frontierswe","benchmarkName":"FrontierSWE","benchmarkCategory":"coding","benchmarkOrganisation":"FrontierSWE","benchmarkVersion":"2026","score":81.2,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-programbench-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-programbench","benchmarkName":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","benchmarkCategory":"coding","benchmarkOrganisation":"John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","benchmarkVersion":"2026","score":77.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-kimicodebenchv2-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"kimi-code-bench-v2","benchmarkName":"Kimi Code Bench v2","benchmarkCategory":"coding","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":72.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-swemarathon-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-marathon","benchmarkName":"SWE Marathon","benchmarkCategory":"coding","benchmarkOrganisation":"Abundant AI","benchmarkVersion":"2026","score":42,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-posttrainbench-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"posttrain-bench","benchmarkName":"PostTrainBench","benchmarkCategory":"coding","benchmarkOrganisation":"PostTrainBench","benchmarkVersion":"2026","score":36.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mlsbenchlite-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mlsbenchlite","benchmarkName":"MLS-Bench Lite","benchmarkCategory":"coding","benchmarkOrganisation":"MLS-Bench","benchmarkVersion":"2026","score":48.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aacodingindex-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76.24,"normalizedScore":98.3386,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aascicode-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":58.7,"normalizedScore":97.4705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-lcr-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.7,"normalizedScore":98.679,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-critpt-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":23.4,"normalizedScore":72.4458,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-officeqapro-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":63.3,"normalizedScore":87.1681,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mmmupro-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81.6,"normalizedScore":60,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mmmupropython-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.4,"normalizedScore":92.053,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-charxivnotools-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":84.8,"normalizedScore":65.5462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-charxiv-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":91.3,"normalizedScore":94.6078,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mathvision-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mathvisionpython-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mathvisionpython","benchmarkName":"MathVision with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI / MathVision authors","benchmarkVersion":"2026","score":97.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-babyvisionpython-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-babyvisionpython","benchmarkName":"BabyVision with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":85.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-zerobench-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"2026","score":23,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-zerobenchpython-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-zerobenchpython","benchmarkName":"ZeroBench_main with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI / ZeroBench authors","benchmarkVersion":"2026","score":41,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-worldvqaforceanswer-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-worldvqaforceanswer","benchmarkName":"WorldVQA ForceAnswer","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI / WorldVQA authors","benchmarkVersion":"2026","score":51,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-omnidocbench-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnidocbench","benchmarkName":"OmniDocBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI / OmniDocBench authors","benchmarkVersion":"2026","score":91.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-perceptionbench-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-perceptionbench","benchmarkName":"PerceptionBench (Internal)","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":58.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aammmupro-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.5,"normalizedScore":93.4256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-designarenawebsite-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1386,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gpqa-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":93.5,"normalizedScore":97.1465,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gpqadiamond-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":93.5,"normalizedScore":97.1465,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-hle-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":56,"normalizedScore":85.0352,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-hlenotools-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":43.5,"normalizedScore":71.1896,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaomniscienceindex-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.4,"normalizedScore":82.8885,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-omniscienceaccuracy-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46,"normalizedScore":73.5395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-omnisciencehallucinationrate-2026-07-21","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.9,"normalizedScore":55.6092,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-terminalbench2-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":80,"normalizedScore":78.8256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-mcpatlas-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":88.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-toolathlon-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":75.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-osworldverified-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":80.8,"normalizedScore":90.8696,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-deepsearchqa-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":84.9,"normalizedScore":68.6335,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-cybergym-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":59,"normalizedScore":36.1556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-financeagentv2-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":57.2,"normalizedScore":83.3123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-deepswe-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":53.3,"normalizedScore":39.9381,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-osworld2-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":14.2,"normalizedScore":19.0635,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-jobbench-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":54.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-cybench-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"cybench","benchmarkName":"Cybench","benchmarkCategory":"agents","benchmarkOrganisation":"Stanford / Cybench authors","benchmarkVersion":"2025","score":92.9,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-exploitgym-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":0.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaagenticindex-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.54,"normalizedScore":69.3653,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-gdpvalaanormalized-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.7,"normalizedScore":70.0321,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-gdpvalaa-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1374,"normalizedScore":80.2326,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aabriefcaseelo-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":863,"normalizedScore":33.427,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaautomationbench-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.8,"normalizedScore":63.4686,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaharveylab-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.1,"normalizedScore":95.7983,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aatau3banking-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.2,"normalizedScore":56.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaterminalbench21-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.9,"normalizedScore":57.9167,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-swepro-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":61.5,"normalizedScore":49.7326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aacodingindex-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.34,"normalizedScore":91.2598,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aascicode-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":58.2,"normalizedScore":96.6273,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-mrcr1m-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":54.1,"normalizedScore":48.3304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-lcr-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.3,"normalizedScore":83.6196,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-critpt-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":15.1,"normalizedScore":46.7492,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-charxiv-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":88.4,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-babyvision-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-babyvision","benchmarkName":"BabyVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":76.3,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-designarenawebsite-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1299,"normalizedScore":85.6436,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-hle-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":62.1,"normalizedScore":95.7746,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-hlenotools-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":52.2,"normalizedScore":87.3606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-healthbenchprofessional-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":59.3,"normalizedScore":90.3226,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aagpqadiamond-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.8,"normalizedScore":94.1734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaomniscienceindex-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18,"normalizedScore":82.5746,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-omniscienceaccuracy-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.6,"normalizedScore":64.2612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-omnisciencehallucinationrate-2026-07-21","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.1,"normalizedScore":71.0495,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-terminalbench2-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":82,"normalizedScore":82.3843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-cybergym-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":81.8,"normalizedScore":88.3295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-browsecomp-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":84.4,"normalizedScore":83.682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-osworldverified-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":78.7,"normalizedScore":86.3043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mcpatlas-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":75.3,"normalizedScore":78.157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-toolathlon-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":55.6,"normalizedScore":58.9322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-tau2bench-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.9,"normalizedScore":94.7528,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaagenticindex-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.87,"normalizedScore":83.0076,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-apexagentsaa-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":37.7,"normalizedScore":79.7414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.5,"normalizedScore":79.3269,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gdpvalaa-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1490,"normalizedScore":86.3636,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gertlabs-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":72.93,"normalizedScore":99.9155,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-researchclawbench-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":17,"normalizedScore":52.8736,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-osworld2-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":13,"normalizedScore":17.0569,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-jobbench-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":42.7,"normalizedScore":74.0091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-exploitgym-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":13.4,"normalizedScore":38.2979,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aabriefcaseelo-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1154,"normalizedScore":60.6742,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaautomationbench-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.1,"normalizedScore":60.8856,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaenterpriseopsgym-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.6,"normalizedScore":82.4219,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaharveylab-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.3,"normalizedScore":76.7507,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaitbench-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.8,"normalizedScore":79.4466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aatau3banking-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.3,"normalizedScore":88.9474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-terminalbenchhard-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":60.6,"normalizedScore":87.0416,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaterminalbench21-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.3,"normalizedScore":84.5833,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-swepro-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":58.6,"normalizedScore":41.9786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-vibecodebench-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":69.847,"normalizedScore":98.3719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-reactnativeevals-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":84.7,"normalizedScore":54.5817,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aacodingindex-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.89,"normalizedScore":96.3883,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aascicode-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":93.086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiercode-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":43,"normalizedScore":64.0411,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mrcrv2-64-128-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mrcrv2-64-128","benchmarkName":"OpenAI MRCR v2 8-needle 64K-128K","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mrcrv2-128-256-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mrcrv2-128-256","benchmarkName":"OpenAI MRCR v2 8-needle 128K-256K","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":87.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-arcagi2-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":85,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-lcr-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.3,"normalizedScore":98.1506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-critpt-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":27.1,"normalizedScore":83.9009,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mmmupro-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81.2,"normalizedScore":58.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mmmupropython-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.2,"normalizedScore":90.7285,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-officeqapro-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":54.1,"normalizedScore":46.4602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aammmupro-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":79.9,"normalizedScore":92.3875,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-designarenawebsite-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1282,"normalizedScore":82.8383,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gpqa-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gpqadiamond-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-hle-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":52.2,"normalizedScore":78.3451,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-hlenotools-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":41.4,"normalizedScore":67.2862,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.1,"normalizedScore":84.2229,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.9,"normalizedScore":92.268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.5,"normalizedScore":13.8721,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaifbench-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.9,"normalizedScore":89.41,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiermath-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":51.7,"normalizedScore":17.4779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":51.7,"normalizedScore":58.0899,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiermathv2tier4-2026-07-21","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":35.4,"normalizedScore":42.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-terminalbench2-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":70.3,"normalizedScore":61.5658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-qwenclawbench-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":61.8,"normalizedScore":80,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-qwenwebbench-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1536,"normalizedScore":81.2865,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-claweval-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.7,"normalizedScore":79.8883,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-bfclv4-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":72.9,"normalizedScore":96.1089,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mcpatlas-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":73.2,"normalizedScore":74.5734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-vitabench-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":45.6,"normalizedScore":92.9012,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-deepplanning-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":62.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-osworldverified-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":73.3,"normalizedScore":74.5652,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-androidworld-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":81,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aaagenticindex-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.81,"normalizedScore":38.2282,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-apexagentsaa-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":22.4,"normalizedScore":46.7672,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-tau2bench-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93,"normalizedScore":93.8446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gdpvalaanormalized-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.8,"normalizedScore":34.9359,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gdpvalaa-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":936,"normalizedScore":57.0825,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-osworld2-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":2.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-sweverified-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.7,"normalizedScore":75.2434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-swepro-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.6,"normalizedScore":39.3048,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-swemultilingual-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":75.8,"normalizedScore":59.204,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-nl2repo-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":41.1,"normalizedScore":64.0553,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-scicode-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":51.3,"normalizedScore":73.4139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-livecodebench-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":89.6,"normalizedScore":96.2963,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aacodingindex-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.86,"normalizedScore":68.8963,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aascicode-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.5,"normalizedScore":75.2108,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-critpt-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":9.1,"normalizedScore":28.1734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mrcrv2-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":91.7,"normalizedScore":96.2151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-lcr-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65,"normalizedScore":85.8653,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmmupro-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79,"normalizedScore":51.6129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mathvision-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":90.3,"normalizedScore":80,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-charxiv-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":85.9,"normalizedScore":81.3725,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-erqa-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":69.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-medxpertqamm-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":71,"normalizedScore":68.4049,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-screenspotpro-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":79,"normalizedScore":78.91,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-simplevqa-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":81.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmsearchplus-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmsearchplus","benchmarkName":"MMSearch-Plus","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":41.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-realworldqa-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-omnidocbench15-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnidocbench15","benchmarkName":"OmniDocBench 1.5","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":91.4,"normalizedScore":88.2353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-ocrbenchv2-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ocrbench-v2","benchmarkName":"OCRBench v2","benchmarkCategory":"multimodal","benchmarkOrganisation":"OCRBench authors","benchmarkVersion":"2025","score":70.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-odinw13-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-odinw13","benchmarkName":"ODINW13","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":51.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-videommewithsub-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-videommmu-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":85.4,"normalizedScore":43.5897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mlvuavg-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mlvuavg","benchmarkName":"MLVU mean average","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aammmupro-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.5,"normalizedScore":93.4256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-designarenawebsite-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1288,"normalizedScore":83.8284,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gpqa-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.3,"normalizedScore":92.581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gpqadiamond-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":90.3,"normalizedScore":92.581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-hle-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":34.7,"normalizedScore":47.5352,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmlupro-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":88.5,"normalizedScore":98.4348,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmluredux-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.5,"normalizedScore":92.0874,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-supergpqa-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":71.4,"normalizedScore":67.1584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmmlu-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":89,"normalizedScore":74.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aaomniscienceindex-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.4,"normalizedScore":70.3297,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-omniscienceaccuracy-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.2,"normalizedScore":32.646,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-omnisciencehallucinationrate-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.5,"normalizedScore":86.2485,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmluprox-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":85.4,"normalizedScore":78.9474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-nova63-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":58.8,"normalizedScore":92.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-include-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-include","benchmarkName":"INCLUDE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-maxife-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-maxife","benchmarkName":"MAXIFE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-polymath-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-polymath","benchmarkName":"PolyMath","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-ifeval-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.6,"normalizedScore":98.818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-ifbench-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":79.1,"normalizedScore":87.3391,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aaifbench-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":78,"normalizedScore":92.587,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-hmmtfeb2026-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.9,"normalizedScore":94.1127,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-imoanswerbench-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":86,"normalizedScore":92.6874,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-apex-2026-07-21","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":22.7,"normalizedScore":50.5669,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-pro-claweval-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.8,"normalizedScore":73.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-deepsearchqa-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":69.7,"normalizedScore":21.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-tau2bench-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.6,"normalizedScore":96.4682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaagenticindex-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.4,"normalizedScore":39.3263,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-apexagentsaa-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":32,"normalizedScore":67.4569,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gdpvalaanormalized-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.3,"normalizedScore":37.3397,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gdpvalaa-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":965,"normalizedScore":58.6152,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gertlabs-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":56.87,"normalizedScore":65.9763,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-researchclawbench-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":13.3,"normalizedScore":10.3448,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaautomationbench-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.5,"normalizedScore":43.9114,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaenterpriseopsgym-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.2,"normalizedScore":65.2344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaharveylab-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaitbench-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.3,"normalizedScore":48.8142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aatau3banking-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.5,"normalizedScore":11.0526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-terminalbenchhard-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":53.8,"normalizedScore":70.4156,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaterminalbench21-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.8,"normalizedScore":40.8333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-livecodebenchpro-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":82.9,"normalizedScore":88.4028,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-reactnativeevals-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":78.9,"normalizedScore":31.4741,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-vibecodebench-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":32.034,"normalizedScore":45.1164,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aacodingindex-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.83,"normalizedScore":87.6336,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aascicode-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":58.9,"normalizedScore":97.8078,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-arcagi2-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":77.1,"normalizedScore":88.9356,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-lcr-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.7,"normalizedScore":96.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-critpt-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":17.7,"normalizedScore":54.7988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-mmmupro-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":83.9,"normalizedScore":67.4194,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-charxiv-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":80.2,"normalizedScore":67.402,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-erqa-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":69.4,"normalizedScore":97.8022,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-simplevqa-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":72.4,"normalizedScore":63.6719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-screenspotpro-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":84.4,"normalizedScore":91.7062,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-zerobench-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"2026","score":29,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-medxpertqamm-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":81.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aammmupro-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":82.4,"normalizedScore":96.7128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-designarenawebsite-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1281,"normalizedScore":82.6733,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gpqadiamond-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":94.3,"normalizedScore":98.2879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-hlenotools-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":45.4,"normalizedScore":74.7212,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-healthbenchhard-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":20.6,"normalizedScore":20.7143,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-medxpertqatext-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":71.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aahle-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":44.7,"normalizedScore":82.8685,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaomniscienceindex-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.9,"normalizedScore":94.27,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-omniscienceaccuracy-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.3,"normalizedScore":89.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.9,"normalizedScore":56.8154,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaglobalmmlulite-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaifbench-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":77.1,"normalizedScore":91.2254,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-frontiermathv2tiers13-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":36.9,"normalizedScore":41.4607,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-frontiermathv2tier4-2026-07-21","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":16.7,"normalizedScore":20.1205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gpt-5-6-luna-terminalbench2-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":84.7,"normalizedScore":87.1886,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-browsecomp-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.3,"normalizedScore":81.3808,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-osworld2-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":45.6,"normalizedScore":71.5719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-cybergym-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":77.9,"normalizedScore":79.405,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-exploitgym-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":12.4,"normalizedScore":35.2584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-toolathlon-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":53.4,"normalizedScore":54.4148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaagenticindex-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.6,"normalizedScore":84.3663,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.2,"normalizedScore":86.859,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gdpvalaa-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1584,"normalizedScore":91.3319,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaharveylab-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.9,"normalizedScore":81.2325,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaitbench-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.3,"normalizedScore":68.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aatau3banking-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.2,"normalizedScore":67.3684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaautomationbench-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.2,"normalizedScore":61.2546,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaterminalbench21-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.9,"normalizedScore":70.4167,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-apexagentsaa-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":35.8,"normalizedScore":75.6466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-swepro-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":62.7,"normalizedScore":52.9412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-deepswe-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":67.2,"normalizedScore":82.9721,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiercode11extended-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode11extended","benchmarkName":"FrontierCode 1.1 Extended","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":55.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aacodingindex-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.45,"normalizedScore":91.4187,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aascicode-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":52.5,"normalizedScore":87.0152,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-arcagi3-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-lcr-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-critpt-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":20.6,"normalizedScore":63.7771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-mmmupro-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.4,"normalizedScore":49.6774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-mmmupropython-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":79.5,"normalizedScore":66.2252,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aammmupro-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.6,"normalizedScore":90.1384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gpqa-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.3,"normalizedScore":95.4344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gpqadiamond-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.3,"normalizedScore":95.4344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-healthbenchprofessional-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":55.7,"normalizedScore":61.2903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-healthbenchhard-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":32,"normalizedScore":61.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aahle-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":37.2,"normalizedScore":67.9283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-11.2,"normalizedScore":59.6546,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.5,"normalizedScore":65.8076,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.1,"normalizedScore":8.3233,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiermath-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":78.6,"normalizedScore":76.9912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":78.6,"normalizedScore":88.3146,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiermathv2tier4-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":58.5,"normalizedScore":70.4819,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-exploitbench-2026-07-21","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-exploitbench","benchmarkName":"ExploitBench v8-bench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Seunghyun Lee, David Brumley, Carnegie Mellon University","benchmarkVersion":"2026","score":33.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-terminalbench2-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":81,"normalizedScore":80.605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-mcpatlas-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":76.8,"normalizedScore":80.7167,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-toolathlon-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":48.2,"normalizedScore":43.7372,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaagenticindex-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.06,"normalizedScore":79.6389,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-tau2bench-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":99.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gdpvalaanormalized-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.7,"normalizedScore":81.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gdpvalaa-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1514,"normalizedScore":87.6321,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-apexagentsaa-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":33.7,"normalizedScore":71.1207,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-researchclawbench-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":20.7,"normalizedScore":95.4023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aabriefcaseelo-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1260,"normalizedScore":70.5993,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaautomationbench-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.8,"normalizedScore":8.1181,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaenterpriseopsgym-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.7,"normalizedScore":67.1875,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaharveylab-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91,"normalizedScore":89.916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaitbench-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.7,"normalizedScore":73.3202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aatau3banking-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":65.2632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-terminalbenchhard-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":50.8,"normalizedScore":63.0807,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaterminalbench21-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.9,"normalizedScore":57.9167,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-swepro-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":62.1,"normalizedScore":51.3369,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-nl2repo-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":48.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-programbench-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-programbench","benchmarkName":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","benchmarkCategory":"coding","benchmarkOrganisation":"John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","benchmarkVersion":"2026","score":63.7,"normalizedScore":41.7355,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aacodingindex-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.76,"normalizedScore":87.5325,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aascicode-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50.5,"normalizedScore":83.6425,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-critpt-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":20.9,"normalizedScore":64.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-lcr-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.3,"normalizedScore":94.1876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-designarenawebsite-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1340,"normalizedScore":92.4092,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gpqa-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":91.2,"normalizedScore":93.865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gpqadiamond-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":91.2,"normalizedScore":93.865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hle-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":54.7,"normalizedScore":82.7465,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hlenotools-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":40.5,"normalizedScore":65.6134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaomniscienceindex-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4,"normalizedScore":71.5856,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-omniscienceaccuracy-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.1,"normalizedScore":37.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-omnisciencehallucinationrate-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.1,"normalizedScore":83.1122,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaopennessindex-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.4,"normalizedScore":22.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaifbench-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.3,"normalizedScore":85.4766,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aime2026-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":99.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hmmtnov2025-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":94.4,"normalizedScore":67.9487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hmmtfeb2026-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.5,"normalizedScore":93.552,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-mmanswerbench-2026-07-21","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":91,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-terminalbench2-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":76.2,"normalizedScore":72.0641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mcpatlas-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":83.6,"normalizedScore":92.3208,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-toolathlon-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":56.5,"normalizedScore":60.7803,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-osworldverified-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":78.4,"normalizedScore":85.6522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-financeagentv2-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":57.861,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gdpvalaa-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1349,"normalizedScore":78.9112,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-tau2bench-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.3,"normalizedScore":96.1655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gdpvalaanormalized-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.4,"normalizedScore":67.9487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaagenticindex-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.45,"normalizedScore":69.1978,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-apexagentsaa-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":47.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gertlabs-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":61.85,"normalizedScore":76.5004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-researchclawbench-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18,"normalizedScore":64.3678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaautomationbench-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.6,"normalizedScore":62.7306,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaenterpriseopsgym-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.1,"normalizedScore":96.0938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-terminalbenchhard-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":40.9,"normalizedScore":38.8753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-swepro-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.1,"normalizedScore":32.6203,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-scicode-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":53.1,"normalizedScore":78.852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-vibecodebench-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":48.683,"normalizedScore":68.5647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aacodingindex-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70.14,"normalizedScore":89.5261,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aascicode-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.1,"normalizedScore":88.027,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mrcrv2-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":77.3,"normalizedScore":67.5299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mrcr1m-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":26.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-arcagi2-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":72.1,"normalizedScore":81.9328,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lcr-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":91.5456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-critpt-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":13.1,"normalizedScore":40.5573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-charxiv-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":84.2,"normalizedScore":77.2059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mmmupro-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":83.6,"normalizedScore":66.4516,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-blueprintbench2-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"blueprint-bench-2","benchmarkName":"Blueprint-Bench 2","benchmarkCategory":"multimodal","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":33.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aammmupro-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":84.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-designarenawebsite-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1285,"normalizedScore":83.3333,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gpqa-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.2,"normalizedScore":95.2918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gpqadiamond-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.676,"normalizedScore":95.9709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-hle-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":40.2,"normalizedScore":57.2183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-omniscienceaccuracy-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.9,"normalizedScore":83.677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.7,"normalizedScore":43.7877,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaomniscienceindex-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.7,"normalizedScore":86.2637,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-ifbench-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":76.3,"normalizedScore":81.3305,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaifbench-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.3,"normalizedScore":90.0151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-frontiermathv2tiers13-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":38.966,"normalizedScore":43.782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-frontiermathv2tier4-2026-07-21","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.583,"normalizedScore":17.5699,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-terminalbench2-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":65.4,"normalizedScore":52.847,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-browsecomp-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.7,"normalizedScore":82.2176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-osworldverified-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":72.7,"normalizedScore":73.2609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-tau2bench-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-claweval-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":70.4,"normalizedScore":90.6425,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-deepsearchqa-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":73.7,"normalizedScore":33.8509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-cybergym-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":66.6,"normalizedScore":53.5469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-gertlabs-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":61.85,"normalizedScore":76.5004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-researchclawbench-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":19.9,"normalizedScore":86.2069,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-jobbench-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":36.7,"normalizedScore":61.0136,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-sweverified-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.84,"normalizedScore":79.6106,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-sweverifiedarcee-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-livecodebenchpro-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":70.7,"normalizedScore":70.4932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-swepro-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":53.4,"normalizedScore":28.0749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-swerebench-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":65.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-reactnativeevals-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":84.1,"normalizedScore":52.1912,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-vibecodebench-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":57.573,"normalizedScore":81.0853,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aascicode-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.7,"normalizedScore":75.5481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-frontiercode-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":26.9,"normalizedScore":8.9041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-lcr-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.3,"normalizedScore":77.0145,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-critpt-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.8,"normalizedScore":8.6687,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-mmmupro-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":77.3,"normalizedScore":46.129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-erqa-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":51.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-screenspotpro-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":83.1,"normalizedScore":88.6256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-medxpertqamm-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":64.8,"normalizedScore":49.3865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aammmupro-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.5,"normalizedScore":79.5848,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-designarenawebsite-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1325,"normalizedScore":89.934,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-gpqa-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":91.3,"normalizedScore":94.0077,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-gpqadiamond-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":89.2,"normalizedScore":91.0116,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-supergpqa-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-mmlupro-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":82,"normalizedScore":89.1861,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-mmluproarcee-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":89.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-hle-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":53,"normalizedScore":79.7535,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-hlenotools-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":40,"normalizedScore":64.684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-healthbenchhard-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":14.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-medxpertqatext-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":52.1,"normalizedScore":8.9202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aaomniscienceindex-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.5,"normalizedScore":71.1931,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-omniscienceaccuracy-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.2,"normalizedScore":72.1649,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-omnisciencehallucinationrate-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76,"normalizedScore":25.3317,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aaifbench-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":44.6,"normalizedScore":42.0575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aime2025arcee-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":99.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-frontiermathv2tiers13-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":40.7,"normalizedScore":45.7303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-frontiermathv2tier4-2026-07-21","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":22.9,"normalizedScore":27.5904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-terminalbench2-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":67.9,"normalizedScore":57.2954,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-browsecomp-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.4,"normalizedScore":81.59,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-hlewithtools-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":48.2,"normalizedScore":54,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-mcpatlas-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":73.6,"normalizedScore":75.256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gdpvalaa-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1307,"normalizedScore":76.6913,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-toolathlon-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":51.8,"normalizedScore":51.1294,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaagenticindex-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":36.36,"normalizedScore":67.1692,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-apexagentsaa-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":24.3,"normalizedScore":50.8621,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-tau2bench-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":96.2,"normalizedScore":97.0737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gdpvalaanormalized-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.4,"normalizedScore":64.7436,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aabriefcaseelo-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":932,"normalizedScore":39.8876,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaenterpriseopsgym-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.4,"normalizedScore":58.2031,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaharveylab-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.4,"normalizedScore":71.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaitbench-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.3,"normalizedScore":64.6245,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aatau3banking-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.8,"normalizedScore":60,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-terminalbenchhard-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":46.2,"normalizedScore":51.8337,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaterminalbench21-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-codeforces-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-codeforces","benchmarkName":"Codeforces Rating","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":3206,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-sweverified-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.6,"normalizedScore":79.2768,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-swepro-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.4,"normalizedScore":33.4225,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-swemultilingual-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":76.2,"normalizedScore":60.199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-vibecodebench-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":49.931,"normalizedScore":70.3224,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aacodingindex-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59.36,"normalizedScore":73.9526,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aascicode-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50,"normalizedScore":82.7993,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-mrcr1m-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":83.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-corpusqa1m-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":62,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-lcr-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.3,"normalizedScore":87.5826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-critpt-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":12.9,"normalizedScore":39.9381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-designarenawebsite-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1264,"normalizedScore":79.868,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-mmlupro-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":87.5,"normalizedScore":97.012,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-simpleqa-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":57.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-chinesesimpleqa-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":84.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gpqa-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.1,"normalizedScore":92.2956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gpqadiamond-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":90.1,"normalizedScore":92.2956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-hle-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":37.7,"normalizedScore":52.8169,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaomniscienceindex-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10,"normalizedScore":60.5965,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-omniscienceaccuracy-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.3,"normalizedScore":68.9003,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-omnisciencehallucinationrate-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94,"normalizedScore":3.6188,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaopennessindex-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50,"normalizedScore":33.4,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaifbench-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.5,"normalizedScore":90.3177,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-hmmtfeb2026-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":95.2,"normalizedScore":97.3367,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-imoanswerbench-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":89.8,"normalizedScore":99.6344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-apex-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":38.3,"normalizedScore":85.941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-apexshortlist-2026-07-21","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-terminalbench2-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":75.1,"normalizedScore":70.1068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-cybergym-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":79,"normalizedScore":81.9222,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-browsecomp-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":82.7,"normalizedScore":80.1255,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-osworldverified-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":75,"normalizedScore":78.2609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mcpatlas-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":70.6,"normalizedScore":70.1365,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-toolathlon-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":54.6,"normalizedScore":56.8789,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-tau2bench-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":87.1,"normalizedScore":87.891,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-claweval-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":60.3,"normalizedScore":76.5363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-deepsearchqa-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":73.6,"normalizedScore":33.5404,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aaagenticindex-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.08,"normalizedScore":75.9538,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-apexagentsaa-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":33.3,"normalizedScore":70.2586,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.7,"normalizedScore":71.6346,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gdpvalaa-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1395,"normalizedScore":81.3425,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gertlabs-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":64.89,"normalizedScore":82.9248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-researchclawbench-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":15.3,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-jobbench-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":38.9,"normalizedScore":65.7786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-exploitgym-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":6,"normalizedScore":15.8055,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-livecodebenchpro-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":87.5,"normalizedScore":95.1556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-swepro-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.7,"normalizedScore":39.5722,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-reactnativeevals-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":85.3,"normalizedScore":56.9721,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-vibecodebench-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":67.421,"normalizedScore":94.9551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aacodingindex-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.05,"normalizedScore":90.8408,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aascicode-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.6,"normalizedScore":93.9292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-lcr-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-critpt-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":23.4,"normalizedScore":72.4458,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mmmupro-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81.2,"normalizedScore":58.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-officeqapro-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":53.2,"normalizedScore":42.4779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mmmupropython-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":82.1,"normalizedScore":83.4437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-charxiv-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":82.8,"normalizedScore":73.7745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-erqa-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":65.4,"normalizedScore":75.8242,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-simplevqa-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":61.1,"normalizedScore":19.5313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-screenspotpro-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":85.4,"normalizedScore":94.0758,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-zerobench-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"2026","score":41,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-medxpertqamm-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":77.1,"normalizedScore":87.1166,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aammmupro-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.4,"normalizedScore":89.7924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-designarenawebsite-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1250,"normalizedScore":77.5578,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gpqa-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.8,"normalizedScore":96.1478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-hle-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":52.1,"normalizedScore":78.169,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-hlenotools-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":39.8,"normalizedScore":64.3123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gpqadiamond-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.8,"normalizedScore":96.1478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-healthbenchhard-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":40.1,"normalizedScore":90.3571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-medxpertqatext-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":59.6,"normalizedScore":44.1315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.7,"normalizedScore":72.9199,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50,"normalizedScore":80.4124,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.6,"normalizedScore":10.1327,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-healthbenchprofessional-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":48.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aaifbench-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.9,"normalizedScore":86.3843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":47.6,"normalizedScore":53.4831,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-frontiermathv2tier4-2026-07-21","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":27.1,"normalizedScore":32.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-terminalbench2-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":66.7,"normalizedScore":55.1601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-browsecomp-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.2,"normalizedScore":81.1715,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-osworldverified-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":73.1,"normalizedScore":74.1304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-toolathlon-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":50,"normalizedScore":47.4333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mcpatlas-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":55.9,"normalizedScore":45.0512,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-claweval-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.3,"normalizedScore":79.3296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-deepsearchqa-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":92.5,"normalizedScore":92.236,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-wideresearch-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":80.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aaagenticindex-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.27,"normalizedScore":55.8347,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-tau2bench-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gdpvalaanormalized-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.5,"normalizedScore":55.2885,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gdpvalaa-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1189,"normalizedScore":70.4545,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-apexagentsaa-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":28.5,"normalizedScore":59.9138,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gertlabs-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":56.82,"normalizedScore":65.8707,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-researchclawbench-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18,"normalizedScore":64.3678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-osworld2-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":4.6,"normalizedScore":3.01,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-terminalbenchhard-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":43.9,"normalizedScore":46.2103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-sweverified-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.2,"normalizedScore":78.7204,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-livecodebenchv6-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":89.6,"normalizedScore":93.9678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-swepro-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":58.6,"normalizedScore":41.9786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-swemultilingual-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":76.7,"normalizedScore":61.4428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-scicode-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":52.2,"normalizedScore":76.1329,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-vibecodebench-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":37.891,"normalizedScore":53.3654,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aacodingindex-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61.77,"normalizedScore":77.4343,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aascicode-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.5,"normalizedScore":88.7015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-lcr-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-critpt-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":8,"normalizedScore":24.7678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mmmupro-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79.4,"normalizedScore":52.9032,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mmmupropython-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":80.1,"normalizedScore":70.1987,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-charxiv-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":80.4,"normalizedScore":67.8922,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mathvision-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.4,"normalizedScore":65.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-vstar-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":96.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aammmupro-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":79.4,"normalizedScore":91.5225,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-designarenawebsite-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1306,"normalizedScore":86.7987,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gpqa-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.5,"normalizedScore":92.8663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gpqadiamond-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":90.5,"normalizedScore":92.8663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-hle-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":34.7,"normalizedScore":47.5352,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aaomniscienceindex-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.4,"normalizedScore":73.4694,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-omniscienceaccuracy-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.8,"normalizedScore":50.8591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-omnisciencehallucinationrate-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.3,"normalizedScore":69.6019,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aaifbench-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76,"normalizedScore":89.5613,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aime2026-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":96.4,"normalizedScore":95.2365,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-hmmtfeb2026-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.7,"normalizedScore":93.8324,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mmanswerbench-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86,"normalizedScore":58.6777,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-frontiermathv2tiers13-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":38.966,"normalizedScore":43.782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-frontiermathv2tier4-2026-07-21","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.58,"normalizedScore":17.5663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-terminalbench2-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":63.5,"normalizedScore":49.4662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-browsecomp-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":68,"normalizedScore":49.3724,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-tau3bench-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.6,"normalizedScore":19.3798,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-mcpatlas-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":71.8,"normalizedScore":72.1843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-cybergym-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":68.7,"normalizedScore":58.3524,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-claweval-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.3,"normalizedScore":79.3296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aaagenticindex-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.87,"normalizedScore":55.0903,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-tau2bench-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":97.7,"normalizedScore":98.5873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gdpvalaanormalized-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.8,"normalizedScore":60.5769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gertlabs-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":60.11,"normalizedScore":72.8233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gdpvalaa-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1257,"normalizedScore":74.0486,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-researchclawbench-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18.2,"normalizedScore":66.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-swepro-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":58.4,"normalizedScore":41.4439,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-nl2repo-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":42.7,"normalizedScore":71.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-swerebench-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":62.7,"normalizedScore":89.0295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-vibecodebench-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":31.456,"normalizedScore":44.3024,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aacodingindex-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.78,"normalizedScore":68.7807,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aascicode-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.8,"normalizedScore":72.344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-lcr-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.3,"normalizedScore":82.2985,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-critpt-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.6,"normalizedScore":14.2415,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-designarenawebsite-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1305,"normalizedScore":86.6337,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gpqadiamond-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":86.2,"normalizedScore":86.7313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-hle-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":52.3,"normalizedScore":78.5211,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aaomniscienceindex-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.9,"normalizedScore":69.9372,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-omniscienceaccuracy-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.2,"normalizedScore":36.0825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-omnisciencehallucinationrate-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.4,"normalizedScore":81.544,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aaifbench-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.3,"normalizedScore":90.0151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aime2026-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.3,"normalizedScore":93.3651,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-hmmtnov2025-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":94,"normalizedScore":62.8205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-hmmtfeb2026-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":82.6,"normalizedScore":79.6748,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-mmanswerbench-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.8,"normalizedScore":40.4959,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-frontiermathv2tiers13-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":33.448,"normalizedScore":37.582,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-frontiermathv2tier4-2026-07-21","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":12.5,"normalizedScore":15.0602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-terminalbench2-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59.1,"normalizedScore":41.637,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-osworldverified-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":72.1,"normalizedScore":71.9565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-claweval-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":67.8,"normalizedScore":87.0112,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-cybergym-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":65.2,"normalizedScore":50.3432,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-tau2bench-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":79.5,"normalizedScore":80.222,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-gertlabs-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":62.92,"normalizedScore":78.7616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-osworld2-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":8.3,"normalizedScore":9.1973,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-jobbench-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":36.9,"normalizedScore":61.4468,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-sweverified-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":79.6,"normalizedScore":77.886,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-swerebench-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":60.7,"normalizedScore":80.5907,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-reactnativeevals-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":80.6,"normalizedScore":38.247,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-vibecodebench-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":51.476,"normalizedScore":72.4983,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aascicode-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.9,"normalizedScore":77.5717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-frontiercode-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":24.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-lcr-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":57.7,"normalizedScore":76.2219,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-critpt-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-charxiv-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":77.4,"normalizedScore":60.5392,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aammmupro-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":70.6,"normalizedScore":76.2976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-designarenawebsite-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1314,"normalizedScore":88.1188,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-gpqa-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":89.9,"normalizedScore":92.0103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-supergpqa-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-mmlupro-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":79.2,"normalizedScore":85.202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-hle-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":49,"normalizedScore":72.7113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aagpqadiamond-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":79.9,"normalizedScore":80.7588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aaomniscienceindex-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-2.9,"normalizedScore":66.1695,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-omniscienceaccuracy-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38,"normalizedScore":59.7938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-omnisciencehallucinationrate-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.9,"normalizedScore":37.5151,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aaifbench-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":41.2,"normalizedScore":36.9138,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-frontiermathv2tiers13-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":32.4,"normalizedScore":36.4045,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-frontiermathv2tier4-2026-07-21","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":8.3,"normalizedScore":10,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-terminalbench2-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":61.6,"normalizedScore":46.0854,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-claweval-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":58.8,"normalizedScore":74.4413,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-qwenclawbench-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":57.2,"normalizedScore":43.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-tau3bench-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.7,"normalizedScore":19.7674,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-vitabench-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":44.3,"normalizedScore":88.8889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-deepplanning-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":41.5,"normalizedScore":56.5762,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-toolathlon-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":39.8,"normalizedScore":26.4887,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mcpatlas-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":48.2,"normalizedScore":31.9113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mcptasks-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.1,"normalizedScore":99.3377,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-wideresearch-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.3,"normalizedScore":68.599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aaagenticindex-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.55,"normalizedScore":50.7724,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-tau2bench-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":97.7,"normalizedScore":98.5873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gdpvalaanormalized-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.8,"normalizedScore":50.9615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gdpvalaa-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1135,"normalizedScore":67.6004,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gertlabs-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":50.6,"normalizedScore":52.7261,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-researchclawbench-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18,"normalizedScore":64.3678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-sweverified-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":78.8,"normalizedScore":76.7733,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-swepro-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.6,"normalizedScore":36.631,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-swemultilingual-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.8,"normalizedScore":54.2289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-livecodebenchv6-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":87.1,"normalizedScore":89.7788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-vibecodebench-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":25.564,"normalizedScore":36.0041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aacodingindex-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.53,"normalizedScore":66.9749,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aascicode-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.7,"normalizedScore":67.1164,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aineedle-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":68.3,"normalizedScore":46.729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-longbenchv2-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":62,"normalizedScore":87.8173,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-lcr-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-critpt-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.9,"normalizedScore":8.9783,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmmu-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":86,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmmupro-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.8,"normalizedScore":50.9677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mathvision-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88,"normalizedScore":68.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-videommmu-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84,"normalizedScore":7.6923,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-screenspotpro-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":68.2,"normalizedScore":53.3175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-charxiv-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":81.5,"normalizedScore":70.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-vstar-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":96.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aammmupro-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78,"normalizedScore":89.1003,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-designarenawebsite-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1249,"normalizedScore":77.3927,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gpqa-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.4,"normalizedScore":92.7236,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-supergpqa-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":71.6,"normalizedScore":67.4367,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmlupro-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":88.5,"normalizedScore":98.4348,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmluredux-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.5,"normalizedScore":92.0874,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-ceval-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":93.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hle-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":28.8,"normalizedScore":37.1479,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aagpqadiamond-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":88.2,"normalizedScore":92.0054,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aaomniscienceindex-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.7,"normalizedScore":70.5651,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-omniscienceaccuracy-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.2,"normalizedScore":39.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-omnisciencehallucinationrate-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32,"normalizedScore":78.4077,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmluprox-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":84.7,"normalizedScore":69.7368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-nova63-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":57.9,"normalizedScore":70,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-ifeval-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.3,"normalizedScore":97.9314,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-ifbench-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":75.8,"normalizedScore":80.2575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aaifbench-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.2,"normalizedScore":88.351,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aime2026-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.3,"normalizedScore":93.3651,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hmmtfeb2025-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":96.7,"normalizedScore":88.2353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hmmtnov2025-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":94.6,"normalizedScore":70.5128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hmmtfeb2026-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.8,"normalizedScore":86.9638,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmanswerbench-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.8,"normalizedScore":40.4959,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-frontiermathv2tiers13-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":26.207,"normalizedScore":29.4461,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-frontiermathv2tier4-2026-07-21","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":8.333,"normalizedScore":10.0398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-terminalbench2-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":56.2,"normalizedScore":36.4769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-claweval-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.7,"normalizedScore":72.905,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-qwenclawbench-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":54.1,"normalizedScore":18.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-tau3bench-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":65.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-deepplanning-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":14.6,"normalizedScore":0.4175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-toolathlon-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":38,"normalizedScore":22.7926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mcpatlas-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":31.1,"normalizedScore":2.7304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mcptasks-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":60.8,"normalizedScore":11.2583,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-wideresearch-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":69.8,"normalizedScore":46.8599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-tau2bench-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.2,"normalizedScore":99.0918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-cybergym-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":43.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-apexagentsaa-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":14.5,"normalizedScore":29.7414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-gertlabs-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":50.99,"normalizedScore":53.5503,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-sweverified-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.8,"normalizedScore":75.3825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-sweverifiedarcee-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":72.8,"normalizedScore":77.4194,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-swepro-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.1,"normalizedScore":32.6203,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-swemultilingual-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.3,"normalizedScore":52.9851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-swerebench-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":62.8,"normalizedScore":89.4515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-reactnativeevals-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":74.8,"normalizedScore":15.1394,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aascicode-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.2,"normalizedScore":76.3912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-longbenchv2-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.8,"normalizedScore":81.7259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aineedle-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":63.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-lcr-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.3,"normalizedScore":83.6196,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-critpt-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2,"normalizedScore":6.192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-designarenawebsite-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1278,"normalizedScore":82.1782,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-gpqa-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":86,"normalizedScore":86.446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-gpqadiamond-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":86,"normalizedScore":86.446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-supergpqa-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":66.8,"normalizedScore":60.757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmlupro-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.7,"normalizedScore":94.4508,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmluproarcee-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":85.8,"normalizedScore":76.259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hle-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":50.4,"normalizedScore":75.1761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aaomniscienceindex-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2,"normalizedScore":70.0157,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-omniscienceaccuracy-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.9,"normalizedScore":40.7216,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-omnisciencehallucinationrate-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34,"normalizedScore":75.9952,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmluprox-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":83.1,"normalizedScore":48.6842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-nova63-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":55.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-ifeval-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":92.6,"normalizedScore":92.9078,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aaifbench-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.3,"normalizedScore":83.9637,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aime2026-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.8,"normalizedScore":94.2157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aime2025arcee-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":93.3,"normalizedScore":91.4248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hmmtfeb2025-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":97.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hmmtnov2025-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":96.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hmmtfeb2026-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.4,"normalizedScore":85.0014,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmanswerbench-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":82.5,"normalizedScore":29.7521,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-frontiermathv2tiers13-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":16.434,"normalizedScore":18.4652,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-frontiermathv2tier4-2026-07-21","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.1,"normalizedScore":2.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-terminalbench2-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":63.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-browsecomp-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":77.1,"normalizedScore":68.41,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-mcpatlas-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":74.1,"normalizedScore":76.1092,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-designarenaagenticwebdev-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-designarenaagenticwebdev","benchmarkName":"Design Arena Agentic Web Dev Elo","benchmarkCategory":"agents","benchmarkOrganisation":"Design Arena / Intelligence","benchmarkVersion":"2026","score":1257,"normalizedScore":50,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aaagenticindex-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.34,"normalizedScore":59.6873,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gdpvalaanormalized-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":36.9,"normalizedScore":59.1346,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gdpvalaa-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1239,"normalizedScore":73.0973,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aabriefcaseelo-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":836,"normalizedScore":30.8989,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aatau3banking-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.7,"normalizedScore":48.9474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-sweverified-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.6,"normalizedScore":75.1043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-swepro-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":54.3,"normalizedScore":30.4813,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aacodingindex-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.06,"normalizedScore":63.4065,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aascicode-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.1,"normalizedScore":76.2226,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-lcr-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.3,"normalizedScore":83.6196,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-critpt-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.4,"normalizedScore":16.7183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-mmmupro-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":73.5,"normalizedScore":33.871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-charxiv-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":82,"normalizedScore":71.8137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-charxivnotools-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":78.1,"normalizedScore":9.2437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aammmupro-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":73.5,"normalizedScore":81.3149,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gpqa-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.9,"normalizedScore":89.1568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gpqadiamond-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87.9,"normalizedScore":89.1568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-hle-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":46,"normalizedScore":67.4296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-hlenotools-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":30,"normalizedScore":46.0967,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aaomniscienceindex-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.1,"normalizedScore":70.0942,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-omniscienceaccuracy-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40,"normalizedScore":63.2302,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-omnisciencehallucinationrate-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.1,"normalizedScore":40.8926,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-ifbench-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":79.8,"normalizedScore":88.8412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aime2026-2026-07-21","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":97.1,"normalizedScore":96.4274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-terminalbench2-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":66,"normalizedScore":53.9146,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-browsecomp-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.52,"normalizedScore":81.841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-osworldverified-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":70.06,"normalizedScore":67.5217,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-mcpatlas-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":74.2,"normalizedScore":76.2799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-claweval-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":74.5,"normalizedScore":96.3687,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaagenticindex-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.36,"normalizedScore":65.308,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-tau2bench-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":88.9,"normalizedScore":89.7074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-gdpvalaanormalized-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.7,"normalizedScore":71.6346,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-gdpvalaa-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1395,"normalizedScore":81.3425,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-gdpvalrubrics-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gdpvalrubrics","benchmarkName":"GDPval rubrics","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":74.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-bankertoolbench-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-bankertoolbench","benchmarkName":"BankerToolBench","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":76.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-researchclawbench-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":19.8,"normalizedScore":85.0575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-osworld2-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":4.6,"normalizedScore":3.01,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aabriefcaseelo-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1110,"normalizedScore":56.5543,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaenterpriseopsgym-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.1,"normalizedScore":25.7813,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaharveylab-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.4,"normalizedScore":82.6331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-terminalbenchhard-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":42.4,"normalizedScore":42.5428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaterminalbench21-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.2,"normalizedScore":5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-sweverified-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.5,"normalizedScore":79.1377,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-swepro-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":59,"normalizedScore":43.0481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-nl2repo-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":42.13,"normalizedScore":68.8018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aacodingindex-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.57,"normalizedScore":72.8113,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aascicode-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.4,"normalizedScore":75.0422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-vibev2-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibev2","benchmarkName":"VIBE V2","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":50.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-svgbench-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-svgbench","benchmarkName":"SVG-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":63.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-kernelbenchhard-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-kernelbenchhard","benchmarkName":"KernelBench Hard","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":28.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-lcr-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-critpt-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.7,"normalizedScore":11.4551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-officeqapro-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":45.1,"normalizedScore":6.6372,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-omnidocbench15-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omnidocbench15","benchmarkName":"OmniDocBench 1.5","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":91.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-mmmupro-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.1,"normalizedScore":48.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-videommmu-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.6,"normalizedScore":23.0769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-videommewithsub-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-designarenawebsite-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1289,"normalizedScore":83.9934,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aammmupro-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.6,"normalizedScore":90.1384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aagpqadiamond-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":92.9,"normalizedScore":98.374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aahle-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":37.1,"normalizedScore":67.7291,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaomniscienceindex-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.4,"normalizedScore":69.5447,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-omniscienceaccuracy-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15,"normalizedScore":20.2749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-omnisciencehallucinationrate-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.1,"normalizedScore":97.5875,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaopennessindex-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaifbench-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":82.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-usamo2026-2026-07-21","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":85.71,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-terminalbench2-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59.3,"normalizedScore":41.9929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-osworldverified-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":66.3,"normalizedScore":59.3478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-osworld-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":"2026","score":66.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-claweval-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":59.6,"normalizedScore":75.5587,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-qwenclawbench-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":52.3,"normalizedScore":4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-tau3bench-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.2,"normalizedScore":17.8295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-vitabench-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":23.3,"normalizedScore":24.0741,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-deepplanning-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":26.4,"normalizedScore":25.0522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-toolathlon-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":43.5,"normalizedScore":34.0862,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mcpatlas-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":42.3,"normalizedScore":21.843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mcptasks-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":71.8,"normalizedScore":84.106,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-wideresearch-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":76.4,"normalizedScore":78.744,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-cybergym-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":50.6,"normalizedScore":16.9336,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-tau2bench-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86.3,"normalizedScore":87.0838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-gertlabs-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":64.23,"normalizedScore":81.53,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-jobbench-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":32.3,"normalizedScore":51.4836,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-sweverified-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.9,"normalizedScore":79.694,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-livecodebenchv6-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":84.8,"normalizedScore":85.9249,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-swepro-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.1,"normalizedScore":37.9679,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-swemultilingual-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":77.5,"normalizedScore":63.4328,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-nl2repo-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":43.2,"normalizedScore":73.7327,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aascicode-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47,"normalizedScore":77.7403,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-longbenchv2-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":64.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aineedle-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-lcr-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.3,"normalizedScore":86.2616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-critpt-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmmupro-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":70.6,"normalizedScore":24.5161,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mathvision-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-charxiv-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":68.5,"normalizedScore":38.7255,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-videommmu-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.4,"normalizedScore":17.9487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-screenspotpro-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":45.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-vstar-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":67,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aammmupro-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":71.2,"normalizedScore":77.3356,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-designarenawebsite-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1277,"normalizedScore":82.0132,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-gpqa-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-supergpqa-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":70.6,"normalizedScore":66.0451,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmlupro-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":89.5,"normalizedScore":99.8577,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmluredux-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":96.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-ceval-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":92.2,"normalizedScore":66.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hle-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":30.8,"normalizedScore":40.669,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aagpqadiamond-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81,"normalizedScore":82.2493,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aaomniscienceindex-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-3.9,"normalizedScore":65.3846,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-omniscienceaccuracy-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.7,"normalizedScore":64.433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-omnisciencehallucinationrate-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.4,"normalizedScore":26.0555,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aammlupro-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.9,"normalizedScore":90,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmluprox-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":85.7,"normalizedScore":82.8947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-nova63-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":56.7,"normalizedScore":40,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-ifeval-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":90.9,"normalizedScore":87.8842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-ifbench-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":58,"normalizedScore":42.0601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aaifbench-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43,"normalizedScore":39.6369,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aime2026-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.1,"normalizedScore":93.0248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hmmtfeb2025-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":92.9,"normalizedScore":32.3529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hmmtnov2025-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":93.3,"normalizedScore":53.8462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hmmtfeb2026-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.3,"normalizedScore":83.4595,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmanswerbench-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84,"normalizedScore":42.1488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-frontiermathv2tiers13-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":20.69,"normalizedScore":23.2472,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-frontiermathv2tier4-2026-07-21","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-terminalbench2-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":52.5,"normalizedScore":29.8932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-browsecomp-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":62,"normalizedScore":36.8201,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-claweval-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":56.8,"normalizedScore":71.648,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-qwenclawbench-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":51.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-tau3bench-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":68.4,"normalizedScore":10.8527,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-vitabench-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":43.7,"normalizedScore":87.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-deepplanning-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":37.6,"normalizedScore":48.4342,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-toolathlon-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":36.3,"normalizedScore":19.3018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mcpatlas-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":46.1,"normalizedScore":28.3276,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mcptasks-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-wideresearch-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74,"normalizedScore":67.1498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-tau2bench-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.6,"normalizedScore":96.4682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gertlabs-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":46.76,"normalizedScore":44.6112,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-researchclawbench-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":14.2,"normalizedScore":20.6897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aaagenticindex-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.85,"normalizedScore":36.4415,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-apexagentsaa-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":15.3,"normalizedScore":31.4655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gdpvalaanormalized-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.1,"normalizedScore":37.0192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gdpvalaa-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":962,"normalizedScore":58.4567,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-sweverified-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":76.2,"normalizedScore":73.1572,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-livecodebenchv6-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":83.6,"normalizedScore":83.9142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-swepro-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":50.9,"normalizedScore":21.3904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aascicode-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42,"normalizedScore":69.3086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aacodingindex-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.21,"normalizedScore":57.8446,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-longbenchv2-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":63.2,"normalizedScore":93.9086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aineedle-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":68.7,"normalizedScore":50.4673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-lcr-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.7,"normalizedScore":86.79,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-critpt-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.7,"normalizedScore":5.2632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmmupro-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79,"normalizedScore":51.6129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mathvision-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88.6,"normalizedScore":71.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-charxiv-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":80.8,"normalizedScore":68.8725,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-videommmu-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.7,"normalizedScore":25.641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-screenspotpro-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":65.6,"normalizedScore":47.1564,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-vstar-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":95.8,"normalizedScore":96.3211,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aammmupro-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":77.3,"normalizedScore":87.8893,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gpqa-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":88.4,"normalizedScore":89.8702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-supergpqa-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":70.4,"normalizedScore":65.7668,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmlupro-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":87.8,"normalizedScore":97.4388,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmluredux-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.9,"normalizedScore":93.5946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-ceval-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":93,"normalizedScore":90.9091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hle-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":28.7,"normalizedScore":36.9718,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aagpqadiamond-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.3,"normalizedScore":93.4959,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aaomniscienceindex-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-29.8,"normalizedScore":45.0549,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-omniscienceaccuracy-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.4,"normalizedScore":48.4536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-omnisciencehallucinationrate-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.1,"normalizedScore":9.5296,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmluprox-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":84.7,"normalizedScore":69.7368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-nova63-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-ifeval-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":92.6,"normalizedScore":92.9078,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aaifbench-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":78.8,"normalizedScore":93.7973,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aime2026-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":93.3,"normalizedScore":89.9626,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hmmtfeb2025-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":94.8,"normalizedScore":60.2941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hmmtnov2025-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":92.7,"normalizedScore":46.1538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hmmtfeb2026-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.9,"normalizedScore":87.104,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmanswerbench-2026-07-21","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":80.9,"normalizedScore":16.5289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-terminalbench2-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":60,"normalizedScore":43.2384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-osworldverified-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":72.1,"normalizedScore":71.9565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-mcpatlas-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":57.7,"normalizedScore":48.1229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-toolathlon-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":42.9,"normalizedScore":32.8542,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-tau2bench-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83.3,"normalizedScore":84.0565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aaagenticindex-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.17,"normalizedScore":55.6486,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-apexagentsaa-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":28.2,"normalizedScore":59.2672,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.6,"normalizedScore":53.8462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-gdpvalaa-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1171,"normalizedScore":69.5032,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-vibecodebench-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":47.969,"normalizedScore":67.5591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aacodingindex-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.08,"normalizedScore":69.2141,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aascicode-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49.9,"normalizedScore":82.6307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-frontiercode-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":27,"normalizedScore":9.2466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-lcr-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":91.5456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-critpt-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":10,"normalizedScore":30.9598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-mmmupro-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":76.6,"normalizedScore":43.871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-mmmupropython-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":78,"normalizedScore":56.2914,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aammmupro-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":73.3,"normalizedScore":80.9689,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-gpqa-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":88,"normalizedScore":89.2995,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-hle-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":41.5,"normalizedScore":59.507,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-hlenotools-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":28.2,"normalizedScore":42.7509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aagpqadiamond-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87.5,"normalizedScore":91.0569,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-18.7,"normalizedScore":53.7677,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.5,"normalizedScore":58.9347,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.8,"normalizedScore":8.6852,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aaifbench-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.3,"normalizedScore":85.4766,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":28.28,"normalizedScore":31.7753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-frontiermathv2tier4-2026-07-21","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.08,"normalizedScore":2.506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-terminalbench2-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":56.9,"normalizedScore":37.7224,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-browsecomp-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":73.2,"normalizedScore":60.251,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-hlewithtools-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":45.1,"normalizedScore":38.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-mcpatlas-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":69,"normalizedScore":67.4061,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gdpvalaa-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1189,"normalizedScore":70.4545,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-toolathlon-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":47.8,"normalizedScore":42.9158,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaagenticindex-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.06,"normalizedScore":57.305,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-tau2bench-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95,"normalizedScore":95.8628,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gdpvalaanormalized-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.4,"normalizedScore":55.1282,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aabriefcaseelo-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":831,"normalizedScore":30.4307,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaenterpriseopsgym-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.6,"normalizedScore":55.0781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaharveylab-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.3,"normalizedScore":62.7451,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaitbench-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.5,"normalizedScore":51.1858,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aatau3banking-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.9,"normalizedScore":44.7368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-terminalbenchhard-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":35.6,"normalizedScore":25.9169,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-codeforces-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-codeforces","benchmarkName":"Codeforces Rating","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":3052,"normalizedScore":60.5128,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-sweverified-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":79,"normalizedScore":77.0515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-swepro-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.6,"normalizedScore":25.9358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-swemultilingual-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.3,"normalizedScore":52.9851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aacodingindex-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.17,"normalizedScore":69.3441,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aascicode-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":44.9,"normalizedScore":74.199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-mrcr1m-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":78.7,"normalizedScore":91.5641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-corpusqa1m-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":60.5,"normalizedScore":96.7742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-lcr-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63,"normalizedScore":83.2232,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-critpt-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":7.1,"normalizedScore":21.9814,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-designarenawebsite-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1238,"normalizedScore":75.5776,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-mmlupro-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.2,"normalizedScore":95.1622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-simpleqa-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":34.1,"normalizedScore":31.6092,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-chinesesimpleqa-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":78.9,"normalizedScore":57.3643,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gpqa-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":88.1,"normalizedScore":89.4421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gpqadiamond-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":88.1,"normalizedScore":89.4421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-hle-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":34.8,"normalizedScore":47.7113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaomniscienceindex-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-22.9,"normalizedScore":50.471,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-omniscienceaccuracy-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.2,"normalizedScore":58.4192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-omnisciencehallucinationrate-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":95.8,"normalizedScore":1.4475,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaopennessindex-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50,"normalizedScore":33.4,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaifbench-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":79.2,"normalizedScore":94.4024,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-hmmtfeb2026-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.8,"normalizedScore":96.776,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-imoanswerbench-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88.4,"normalizedScore":97.075,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-apex-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":33,"normalizedScore":73.9229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-apexshortlist-2026-07-21","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.7,"normalizedScore":94.4444,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-pro-tau2bench-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":87.1,"normalizedScore":87.891,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-gertlabs-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":63.23,"normalizedScore":79.4167,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-jobbench-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":11.4,"normalizedScore":6.2162,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-vibecodebench-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":14.3,"normalizedScore":20.14,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aascicode-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":93.086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aalivecodebench-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aalivecodebench","benchmarkName":"Artificial Analysis LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-arcagi2-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":31.1,"normalizedScore":24.5098,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-lcr-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70.7,"normalizedScore":93.395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-critpt-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":9.1,"normalizedScore":28.1734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-mmmupro-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81,"normalizedScore":58.0645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-mathvision-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.6,"normalizedScore":61.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-videommmu-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":87.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-screenspotpro-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":72.7,"normalizedScore":63.981,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-charxiv-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":81.4,"normalizedScore":70.3431,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-vstar-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":88,"normalizedScore":70.2341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aammmupro-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.2,"normalizedScore":92.9066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aagpqadiamond-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":90.8,"normalizedScore":95.5285,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aahle-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":37.2,"normalizedScore":67.9283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aaomniscienceindex-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.8,"normalizedScore":80.8477,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-omniscienceaccuracy-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.9,"normalizedScore":90.5498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.9,"normalizedScore":7.3583,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aammlupro-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aaglobalmmlulite-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":92.2,"normalizedScore":90.3846,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aaifbench-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.4,"normalizedScore":81.0893,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-frontiermathv2tiers13-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":37.6,"normalizedScore":42.2472,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-frontiermathv2tier4-2026-07-21","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":18.75,"normalizedScore":22.5904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-kimi-k2-5-terminalbench2-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":50.8,"normalizedScore":26.8683,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-browsecomp-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":60.6,"normalizedScore":33.8912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-claweval-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":52.3,"normalizedScore":65.3631,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-qwenclawbench-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":54.3,"normalizedScore":20,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-tau3bench-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":65.7,"normalizedScore":0.3876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-deepsearchqa-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":77.1,"normalizedScore":44.4099,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-deepplanning-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":14.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-toolathlon-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":27.8,"normalizedScore":1.848,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mcpatlas-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":29.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mcptasks-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-wideresearch-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":72.7,"normalizedScore":60.8696,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-tau2bench-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-apexagentsaa-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":11.5,"normalizedScore":23.2759,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gertlabs-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":45.88,"normalizedScore":42.7515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-researchclawbench-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":14,"normalizedScore":18.3908,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-jobbench-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":8.73,"normalizedScore":0.4332,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aaagenticindex-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.69,"normalizedScore":39.866,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gdpvalaanormalized-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.4,"normalizedScore":40.7051,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gdpvalaa-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1009,"normalizedScore":60.9408,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-sweverified-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":76.8,"normalizedScore":73.9917,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-sweverifiedarcee-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":70.8,"normalizedScore":61.2903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-livecodebenchv6-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":85,"normalizedScore":86.2601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-swepro-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":50.7,"normalizedScore":20.8556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-swemultilingual-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73,"normalizedScore":52.2388,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-swerebench-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.5,"normalizedScore":71.308,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-reactnativeevals-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":77.2,"normalizedScore":24.7012,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-scicode-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":48.7,"normalizedScore":65.5589,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aascicode-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49,"normalizedScore":81.113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aacodingindex-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.78,"normalizedScore":55.7787,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-longbenchv2-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":61,"normalizedScore":82.7411,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-lcr-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.3,"normalizedScore":86.2616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-critpt-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.1,"normalizedScore":9.5975,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmmupro-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-videomme-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-videomme","benchmarkName":"Video-MME","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MME benchmark team","benchmarkVersion":"2024","score":87.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmvu-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":80.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-videommmu-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":86.6,"normalizedScore":74.359,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aammmupro-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.4,"normalizedScore":84.6021,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-designarenawebsite-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1279,"normalizedScore":82.3432,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gpqa-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.6,"normalizedScore":88.7288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gpqadiamond-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87.6,"normalizedScore":88.7288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-supergpqa-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":69.2,"normalizedScore":64.0969,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmlupro-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":87.1,"normalizedScore":96.4428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmluproarcee-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":87.1,"normalizedScore":85.6115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hle-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":30.1,"normalizedScore":39.4366,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aaomniscienceindex-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-8.1,"normalizedScore":62.0879,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-omniscienceaccuracy-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.3,"normalizedScore":53.4364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-omnisciencehallucinationrate-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.6,"normalizedScore":39.0832,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmluprox-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":82.3,"normalizedScore":38.1579,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-nova63-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":56,"normalizedScore":22.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-ifeval-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":93.9,"normalizedScore":96.7494,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aaifbench-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.2,"normalizedScore":80.7867,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aime2025-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":96.1,"normalizedScore":98.4093,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aime2026-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.8,"normalizedScore":94.2157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aime2025arcee-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":96.3,"normalizedScore":95.3826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hmmtfeb2025-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":95.4,"normalizedScore":69.1176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hmmtnov2025-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":91.1,"normalizedScore":25.641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hmmtfeb2026-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.1,"normalizedScore":85.9826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmanswerbench-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.8,"normalizedScore":23.9669,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-frontiermathv2tiers13-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":27.9,"normalizedScore":31.3483,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-frontiermathv2tier4-2026-07-21","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.2,"normalizedScore":5.0602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-browsecomp-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":65.8,"normalizedScore":44.7699,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-osworldverified-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":47.3,"normalizedScore":18.0435,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-tau2bench-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-gertlabs-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":46.54,"normalizedScore":44.1462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-jobbench-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":34.3,"normalizedScore":55.8155,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-sweverified-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80,"normalizedScore":78.4423,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-swepro-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.6,"normalizedScore":33.9572,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-vibecodebench-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":53.499,"normalizedScore":75.3475,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aascicode-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":52.1,"normalizedScore":86.3406,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-arcagi2-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":52.9,"normalizedScore":55.042,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-lcr-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.7,"normalizedScore":96.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-critpt-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":11.6,"normalizedScore":35.9133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-mmmupro-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79.5,"normalizedScore":53.2258,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-mathvision-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83,"normalizedScore":43.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-charxiv-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":82.1,"normalizedScore":72.0588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-vstar-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":75.9,"normalizedScore":29.7659,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-designarenawebsite-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1224,"normalizedScore":73.2673,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-gpqa-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.4,"normalizedScore":95.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aagpqadiamond-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":90.3,"normalizedScore":94.8509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aahle-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":35.4,"normalizedScore":64.3426,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-1,"normalizedScore":67.6609,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.8,"normalizedScore":69.7595,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":79.7,"normalizedScore":20.8685,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aaifbench-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.4,"normalizedScore":88.6536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aaaime2025-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaaime2025","benchmarkName":"Artificial Analysis AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":40.7,"normalizedScore":45.7303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-frontiermathv2tier4-2026-07-21","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":18.8,"normalizedScore":22.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-terminalbench2-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":49.4,"normalizedScore":24.3772,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-browsecomp-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":63.8,"normalizedScore":40.5858,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-osworldverified-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":58,"normalizedScore":41.3043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-tau2bench-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.6,"normalizedScore":94.4501,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aaagenticindex-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.72,"normalizedScore":38.0607,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-gdpvalaanormalized-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.9,"normalizedScore":38.3013,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-gdpvalaa-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":978,"normalizedScore":59.3023,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-sweverified-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":72,"normalizedScore":67.3157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aacodingindex-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.71,"normalizedScore":54.2329,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aascicode-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42,"normalizedScore":69.3086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-longbenchv2-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.2,"normalizedScore":78.6802,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-lcr-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-critpt-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmmu-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":83.9,"normalizedScore":96.0623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmvu-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":74.7,"normalizedScore":29.6296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mathvision-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":59.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-charxiv-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":77.2,"normalizedScore":60.049,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-vstar-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":93.2,"normalizedScore":87.6254,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aammmupro-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75,"normalizedScore":83.91,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmlupro-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.7,"normalizedScore":95.8736,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-supergpqa-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":67.1,"normalizedScore":61.1745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-gpqa-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":86.6,"normalizedScore":87.302,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aagpqadiamond-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.7,"normalizedScore":88.6179,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aahle-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.4,"normalizedScore":40.4382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aaomniscienceindex-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-39.6,"normalizedScore":37.3626,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-omniscienceaccuracy-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.7,"normalizedScore":36.9416,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-omnisciencehallucinationrate-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.5,"normalizedScore":13.8721,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmluprox-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":82.2,"normalizedScore":36.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-ifeval-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":93.4,"normalizedScore":95.2719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aaifbench-2026-07-21","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.7,"normalizedScore":89.1074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-terminalbench2-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":56.4,"normalizedScore":36.8327,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-pinchbench-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-pinchbench","benchmarkName":"PinchBench","benchmarkCategory":"agents","benchmarkOrganisation":"Kilo Code","benchmarkVersion":"2026","score":90,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-browsecomp-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":44.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-tau3bench-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.9,"normalizedScore":20.5426,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gdpvalaanormalized-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.2,"normalizedScore":53.2051,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-hlewithtools-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":37.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaagenticindex-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.36,"normalizedScore":50.4188,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-tau2bench-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83.3,"normalizedScore":84.0565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gdpvalaa-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1164,"normalizedScore":69.1332,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aabriefcaseelo-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":870,"normalizedScore":34.0824,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaenterpriseopsgym-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.9,"normalizedScore":13.2812,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaharveylab-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.7,"normalizedScore":63.8655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-terminalbenchhard-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":36.4,"normalizedScore":27.8729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-sweverified-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":71.9,"normalizedScore":67.1766,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-swemultilingual-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":67.7,"normalizedScore":39.0547,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-scicode-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":44.6,"normalizedScore":53.1722,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aacodingindex-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.27,"normalizedScore":59.3759,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aascicode-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.9,"normalizedScore":65.7673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-lcr-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67,"normalizedScore":88.5073,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-critpt-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.1,"normalizedScore":9.5975,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-longbenchv2-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":61.9,"normalizedScore":87.3096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-designarenawebsite-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1129,"normalizedScore":57.5908,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gpqa-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gpqadiamond-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-hle-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":26.7,"normalizedScore":33.4507,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-hlenotools-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":26.7,"normalizedScore":39.9628,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-mmlupro-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.8,"normalizedScore":96.0159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-omniscienceaccuracy-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.6,"normalizedScore":31.6151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaomniscienceindex-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-0.8,"normalizedScore":67.8179,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-omnisciencehallucinationrate-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.5,"normalizedScore":82.6297,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaopennessindex-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.3,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-mmluprox-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":83,"normalizedScore":47.3684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-ifbench-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":81.7,"normalizedScore":92.9185,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaifbench-2026-07-21","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":81.4,"normalizedScore":97.7307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-terminalbench2-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":41.6,"normalizedScore":10.4982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-browsecomp-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":61,"normalizedScore":34.728,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-osworldverified-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":56.2,"normalizedScore":37.3913,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-tau2bench-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.9,"normalizedScore":94.7528,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-gertlabs-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.41,"normalizedScore":29.0786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-sweverified-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":72.4,"normalizedScore":67.872,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-swerebench-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.9,"normalizedScore":72.9958,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aascicode-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.5,"normalizedScore":65.0927,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-longbenchv2-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.6,"normalizedScore":80.7107,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-lcr-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.3,"normalizedScore":88.9036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-critpt-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmmu-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":82.3,"normalizedScore":93.0621,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmvu-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":73.3,"normalizedScore":12.3457,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mathvision-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86,"normalizedScore":58.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-vstar-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":93.7,"normalizedScore":89.2977,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aammmupro-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75,"normalizedScore":83.91,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmlupro-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.1,"normalizedScore":95.0199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-supergpqa-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":65.6,"normalizedScore":59.0871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-gpqa-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":85.5,"normalizedScore":85.7326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aagpqadiamond-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.8,"normalizedScore":88.7534,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aahle-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":22.2,"normalizedScore":38.0478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aaomniscienceindex-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-42,"normalizedScore":35.4788,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-omniscienceaccuracy-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21,"normalizedScore":30.5842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-omnisciencehallucinationrate-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":79.7,"normalizedScore":20.8685,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmluprox-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":82.2,"normalizedScore":36.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-ifeval-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aaifbench-2026-07-21","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.6,"normalizedScore":88.9561,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-terminalbench2-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":46,"normalizedScore":18.3274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-sweverified-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.5,"normalizedScore":69.4019,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-swepro-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.8,"normalizedScore":26.4706,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-graphwalksbfs128k-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-graphwalksbfs128k","benchmarkName":"Graphwalks BFS 0K-128K","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-gpqa-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":84.2,"normalizedScore":83.8779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-gpqadiamond-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":84.2,"normalizedScore":83.8779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-mmlupro-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85,"normalizedScore":93.4548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-simpleqa-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":31,"normalizedScore":22.7011,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-ifbench-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":85,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-aime2025-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":97,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-aime2026-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":94.5,"normalizedScore":92.0041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-hmmtfeb2026-2026-07-21","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84.9,"normalizedScore":82.8988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-terminalbench2-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59.3,"normalizedScore":41.9929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-claweval-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":72.4,"normalizedScore":93.4358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-qwenclawbench-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":53.4,"normalizedScore":12.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-qwenwebbench-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1487,"normalizedScore":52.6316,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-androidworld-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":70.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aaagenticindex-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.03,"normalizedScore":49.8046,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-tau2bench-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.2,"normalizedScore":95.0555,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gdpvalaanormalized-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32,"normalizedScore":51.2821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gdpvalaa-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1140,"normalizedScore":67.8647,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gertlabs-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":54.84,"normalizedScore":61.6864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-sweverified-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.2,"normalizedScore":74.548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-swemultilingual-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":71.3,"normalizedScore":48.01,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-swepro-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":53.5,"normalizedScore":28.3422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-livecodebench-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":83.9,"normalizedScore":85.7407,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-nl2repo-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":36.2,"normalizedScore":41.4747,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aacodingindex-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":53.72,"normalizedScore":65.8047,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aascicode-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.8,"normalizedScore":65.5987,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-lcr-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.7,"normalizedScore":90.753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-critpt-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmmu-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":82.9,"normalizedScore":94.1871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmmupro-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":75.8,"normalizedScore":41.2903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-realworldqa-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84.1,"normalizedScore":90.1651,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-dynamath-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-dynamath","benchmarkName":"DynaMath","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mstar-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mstar","benchmarkName":"MStar","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-simplevqa-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":56.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-charxiv-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":78.4,"normalizedScore":62.9902,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-ccocr-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ccocr","benchmarkName":"CC-OCR","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-countbench-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-countbench","benchmarkName":"CountBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":97.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-refcocoavg-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":92.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-erqa-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":62.5,"normalizedScore":59.8901,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-videommewithsub-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.7,"normalizedScore":88.4615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-videommmu-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.4,"normalizedScore":17.9487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mlvuavg-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mlvuavg","benchmarkName":"MLVU mean average","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.6,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-vstar-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":94.7,"normalizedScore":92.6421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aammmupro-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.6,"normalizedScore":83.218,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmlupro-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.2,"normalizedScore":95.1622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmluredux-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":93.5,"normalizedScore":88.3195,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-supergpqa-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":66,"normalizedScore":59.6438,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-ceval-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":91.4,"normalizedScore":42.4242,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gpqa-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.8,"normalizedScore":89.0141,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hle-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":24,"normalizedScore":28.6972,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aagpqadiamond-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.2,"normalizedScore":86.5854,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aaomniscienceindex-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-19.8,"normalizedScore":52.9042,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-omniscienceaccuracy-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.2,"normalizedScore":27.4914,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-omnisciencehallucinationrate-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.3,"normalizedScore":58.7455,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aaifbench-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":67.6,"normalizedScore":76.8533,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hmmtfeb2025-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":93.8,"normalizedScore":45.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hmmtnov2025-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":90.7,"normalizedScore":20.5128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hmmtfeb2026-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84.3,"normalizedScore":82.0578,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmanswerbench-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":80.8,"normalizedScore":15.7025,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aime2026-2026-07-21","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":94.1,"normalizedScore":91.3236,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-terminalbench2-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59.1,"normalizedScore":41.637,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-mcpatlas-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":69.4,"normalizedScore":68.0887,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-toolathlon-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":46.3,"normalizedScore":39.8357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-claweval-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":59.8,"normalizedScore":75.838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-gertlabs-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":50.28,"normalizedScore":52.0499,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-researchclawbench-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":17.1,"normalizedScore":54.023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-sweverified-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.6,"normalizedScore":69.541,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-swepro-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.1,"normalizedScore":24.5989,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-swemultilingual-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":69.8,"normalizedScore":44.2786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-mrcr1m-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":44.7,"normalizedScore":31.8102,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-corpusqa1m-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":35.6,"normalizedScore":43.2258,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-designarenawebsite-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1264,"normalizedScore":79.868,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-mmlupro-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":82.9,"normalizedScore":90.4667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-simpleqa-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":45,"normalizedScore":62.931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-chinesesimpleqa-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":75.8,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-gpqa-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":72.9,"normalizedScore":67.7557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-gpqadiamond-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":72.9,"normalizedScore":67.7557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-hle-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":7.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-hmmtfeb2026-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":31.7,"normalizedScore":8.3263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-imoanswerbench-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":35.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-apex-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":0.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-apexshortlist-2026-07-21","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":9.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-terminalbench2-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":40.5,"normalizedScore":8.5409,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-browsecomp-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":61,"normalizedScore":34.728,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-osworldverified-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":54.5,"normalizedScore":33.6957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-tau2bench-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":89.2,"normalizedScore":90.0101,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-gertlabs-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":28.96,"normalizedScore":6.9949,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-sweverified-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":69.2,"normalizedScore":63.4214,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-swerebench-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":53.7,"normalizedScore":51.0549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aascicode-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.7,"normalizedScore":62.0573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-longbenchv2-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":59,"normalizedScore":72.5888,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-lcr-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.7,"normalizedScore":82.8269,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-critpt-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmmu-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":81.4,"normalizedScore":91.3745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmvu-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":72.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mathvision-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.9,"normalizedScore":48,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-vstar-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":92.7,"normalizedScore":85.9532,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aammmupro-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.7,"normalizedScore":79.9308,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmlupro-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.3,"normalizedScore":93.8816,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-supergpqa-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":63.4,"normalizedScore":56.0256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-gpqa-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":84.2,"normalizedScore":83.8779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aagpqadiamond-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.5,"normalizedScore":86.9919,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aahle-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":19.7,"normalizedScore":33.0677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aaomniscienceindex-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-46.4,"normalizedScore":32.0251,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-omniscienceaccuracy-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.5,"normalizedScore":29.7251,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-omnisciencehallucinationrate-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84,"normalizedScore":15.6815,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmluprox-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":81,"normalizedScore":21.0526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-ifeval-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":91.9,"normalizedScore":90.8392,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aaifbench-2026-07-21","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.5,"normalizedScore":84.2663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-terminalbench2-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":51.5,"normalizedScore":28.1139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-claweval-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":68.7,"normalizedScore":88.2682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-qwenclawbench-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":52.6,"normalizedScore":6.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-qwenwebbench-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1397,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-tau3bench-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":67.2,"normalizedScore":6.2016,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-vitabench-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":35.6,"normalizedScore":62.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-deepplanning-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":25.9,"normalizedScore":24.0084,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-toolathlon-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":26.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mcpatlas-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":62.8,"normalizedScore":56.8259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-wideresearch-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":60.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aaagenticindex-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.41,"normalizedScore":39.3449,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-tau2bench-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.3,"normalizedScore":96.1655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gdpvalaanormalized-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.4,"normalizedScore":43.9103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gdpvalaa-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1049,"normalizedScore":63.055,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gertlabs-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":42.65,"normalizedScore":35.9256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-sweverified-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.4,"normalizedScore":69.2629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-swemultilingual-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":67.2,"normalizedScore":37.8109,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-swepro-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":49.5,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-livecodebench-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":80.4,"normalizedScore":79.2593,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-nl2repo-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":29.4,"normalizedScore":10.1382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aacodingindex-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.88,"normalizedScore":48.6998,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aascicode-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.8,"normalizedScore":58.8533,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-lcr-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.7,"normalizedScore":84.148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-critpt-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmmu-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":81.7,"normalizedScore":91.937,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmmupro-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":75.3,"normalizedScore":39.6774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-realworldqa-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.3,"normalizedScore":94.38,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-omnidocbench15-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnidocbench15","benchmarkName":"OmniDocBench 1.5","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":89.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-charxiv-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":78,"normalizedScore":62.0098,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-simplevqa-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":58.9,"normalizedScore":10.9375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-ccocr-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ccocr","benchmarkName":"CC-OCR","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-ai2dtest-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ai2dtest","benchmarkName":"AI2D test split","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-refcocoavg-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":92,"normalizedScore":95.1923,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-odinw13-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-odinw13","benchmarkName":"ODINW13","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":50.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-videommewithsub-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.6,"normalizedScore":46.1538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-videommenosub-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommenosub","benchmarkName":"Video-MME without subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":82.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-videommmu-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":83.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mlvuavg-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mlvuavg","benchmarkName":"MLVU mean average","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aammmupro-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75,"normalizedScore":83.91,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmlupro-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.2,"normalizedScore":93.7393,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-supergpqa-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":64.7,"normalizedScore":57.8347,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-ceval-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":90,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gpqa-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":86,"normalizedScore":86.446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hle-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":21.4,"normalizedScore":24.1197,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aagpqadiamond-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.1,"normalizedScore":86.4499,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aaomniscienceindex-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-21.4,"normalizedScore":51.6484,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-omniscienceaccuracy-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.9,"normalizedScore":26.9759,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-omnisciencehallucinationrate-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.7,"normalizedScore":57.0567,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aaifbench-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":64.4,"normalizedScore":72.0121,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2025-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":90.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hmmtnov2025-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":89.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2026-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.6,"normalizedScore":81.0765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmanswerbench-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":78.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aime2026-2026-07-21","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":92.7,"normalizedScore":88.9418,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-terminalbench2-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":41,"normalizedScore":9.4306,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-browsecomp-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":52,"normalizedScore":15.8996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-vitabench-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":15.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aaagenticindex-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.39,"normalizedScore":46.7523,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-tau2bench-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gertlabs-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.95,"normalizedScore":30.2198,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gdpvalaanormalized-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":53.3654,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gdpvalaa-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1165,"normalizedScore":69.186,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-sweverified-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.8,"normalizedScore":69.8192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-livecodebench-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":84.9,"normalizedScore":87.5926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-swerebench-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.7,"normalizedScore":72.1519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aacodingindex-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.26,"normalizedScore":53.5828,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aascicode-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.1,"normalizedScore":74.5363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aalivecodebench-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aalivecodebench","benchmarkName":"Artificial Analysis LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.4,"normalizedScore":41.0256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-lcr-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64,"normalizedScore":84.5443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-critpt-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.7,"normalizedScore":5.2632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-designarenawebsite-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1255,"normalizedScore":78.3828,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gpqa-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":85.7,"normalizedScore":86.018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-mmlupro-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":84.3,"normalizedScore":92.4587,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-hle-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":24.8,"normalizedScore":30.1056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aagpqadiamond-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.9,"normalizedScore":88.8889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aaomniscienceindex-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-34.6,"normalizedScore":41.2873,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-omniscienceaccuracy-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.3,"normalizedScore":44.8454,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-omnisciencehallucinationrate-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.3,"normalizedScore":8.082,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aaifbench-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":67.9,"normalizedScore":77.3071,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aime2025-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":95.7,"normalizedScore":97.7024,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-frontiermathv2tiers13-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.439,"normalizedScore":2.7404,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-frontiermathv2tier4-2026-07-21","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-terminalbench2-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":46.3,"normalizedScore":18.8612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-osworldverified-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":39,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-mcpatlas-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":56.1,"normalizedScore":45.3925,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-toolathlon-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":35.5,"normalizedScore":17.6591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-tau2bench-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":76,"normalizedScore":76.6902,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aaagenticindex-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.54,"normalizedScore":50.7538,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-apexagentsaa-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":24.9,"normalizedScore":52.1552,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30,"normalizedScore":48.0769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-gdpvalaa-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1100,"normalizedScore":65.7505,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-vibecodebench-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":26.097,"normalizedScore":36.7548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aacodingindex-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.07,"normalizedScore":69.1997,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aascicode-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.9,"normalizedScore":77.5717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-lcr-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66,"normalizedScore":87.1863,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-critpt-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":9.3,"normalizedScore":28.7926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-mmmupro-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":66.1,"normalizedScore":10,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-mmmupropython-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":69.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aammmupro-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":65.4,"normalizedScore":67.301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-gpqa-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":82.8,"normalizedScore":81.8804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-hle-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":37.7,"normalizedScore":52.8169,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-hlenotools-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":24.3,"normalizedScore":35.5019,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aagpqadiamond-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81.7,"normalizedScore":83.1978,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-29.5,"normalizedScore":45.2904,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.4,"normalizedScore":38.1443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.6,"normalizedScore":28.2268,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aaifbench-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.9,"normalizedScore":89.41,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":25.86,"normalizedScore":29.0562,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-frontiermathv2tier4-2026-07-21","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":6.25,"normalizedScore":7.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-terminalbench2-2026-07-21","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":82.1,"normalizedScore":82.5623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-swepro-2026-07-21","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":73.7,"normalizedScore":82.3529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-livecodebenchv6-2026-07-21","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":93.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-livecodebenchpro-2026-07-21","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":90.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-scicode-2026-07-21","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":58.7,"normalizedScore":95.7704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-mrcrv2-2026-07-21","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":93.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-charxiv-2026-07-21","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":86.6,"normalizedScore":83.0882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-gpqa-2026-07-21","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-gpqadiamond-2026-07-21","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-hlenotools-2026-07-21","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":50,"normalizedScore":83.2714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-osworldverified-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":83,"normalizedScore":95.6522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-gdpvalaa-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1421,"normalizedScore":82.7167,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aaagenticindex-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.72,"normalizedScore":71.5615,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-gdpvalaanormalized-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.1,"normalizedScore":73.8782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aabriefcaseelo-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":961,"normalizedScore":42.603,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aatau3banking-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.5,"normalizedScore":53.1579,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aaterminalbench21-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.5,"normalizedScore":56.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-deepswe-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":49,"normalizedScore":26.6254,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aacodingindex-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.24,"normalizedScore":88.2259,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aascicode-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":52.7,"normalizedScore":87.3524,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-lcr-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-critpt-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":10.6,"normalizedScore":32.8173,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aammmupro-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":83.2,"normalizedScore":98.0969,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aagpqadiamond-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":92.8,"normalizedScore":98.2385,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aahle-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":38.3,"normalizedScore":70.1195,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aaomniscienceindex-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.5,"normalizedScore":86.8917,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-omniscienceaccuracy-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.2,"normalizedScore":80.756,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":53.5,"normalizedScore":52.4729,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-terminalbench2-2026-07-21","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":80.2,"normalizedScore":79.1815,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-swepro-2026-07-21","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":59,"normalizedScore":43.0481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-livecodebenchv6-2026-07-21","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":92.9,"normalizedScore":99.4973,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-livecodebenchpro-2026-07-21","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":87.8,"normalizedScore":95.596,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-scicode-2026-07-21","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":60.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-mrcrv2-2026-07-21","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":86.6,"normalizedScore":86.0558,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-charxiv-2026-07-21","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":85.1,"normalizedScore":79.4118,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-gpqa-2026-07-21","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-gpqadiamond-2026-07-21","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-hlenotools-2026-07-21","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":47.2,"normalizedScore":78.0669,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-swe-1-7-terminalbench2-2026-07-21","modelSlug":"swe-1-7","modelName":"SWE-1.7","providerId":"cognition","providerName":"Cognition","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":81.5,"normalizedScore":81.4947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-swe-1-7-frontiercode-2026-07-21","modelSlug":"swe-1-7","modelName":"SWE-1.7","providerId":"cognition","providerName":"Cognition","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":42.3,"normalizedScore":61.6438,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-swe-1-7-swemultilingual-2026-07-21","modelSlug":"swe-1-7","modelName":"SWE-1.7","providerId":"cognition","providerName":"Cognition","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":77.8,"normalizedScore":64.1791,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-35b-a3b-osworldverified-2026-07-21","modelSlug":"holo3-35b-a3b","modelName":"Holo3-35B-A3B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":82.56,"normalizedScore":94.6957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-terminalbench2-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":83.3,"normalizedScore":84.6975,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-deepswe-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":53,"normalizedScore":39.0093,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaagenticindex-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.69,"normalizedScore":84.5338,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-gdpvalaanormalized-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.7,"normalizedScore":82.8526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-gdpvalaa-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1535,"normalizedScore":88.7421,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aabriefcaseelo-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1323,"normalizedScore":76.4981,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaautomationbench-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.4,"normalizedScore":95.203,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaenterpriseopsgym-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.8,"normalizedScore":59.7656,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaharveylab-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":92.4,"normalizedScore":93.8375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aatau3banking-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.6,"normalizedScore":95.7895,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaterminalbench21-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.6,"normalizedScore":73.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-swepro-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":64.7,"normalizedScore":58.2888,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-swemultilingual-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78,"normalizedScore":64.6766,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-vulcanbench-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vulcanbench","benchmarkName":"VulcanBench v3","benchmarkCategory":"coding","benchmarkOrganisation":"VulcanBench contributors","benchmarkVersion":"2026","score":91.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aacodingindex-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.45,"normalizedScore":92.8633,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aascicode-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":54.1,"normalizedScore":89.7133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-lcr-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.7,"normalizedScore":89.432,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-critpt-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":15.4,"normalizedScore":47.678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aammmupro-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.4,"normalizedScore":93.2526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-designarenawebsite-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1325,"normalizedScore":89.934,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aagpqadiamond-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":93.1,"normalizedScore":98.645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aahle-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":40.3,"normalizedScore":74.1036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaomniscienceindex-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.4,"normalizedScore":89.168,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-omniscienceaccuracy-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.1,"normalizedScore":84.0206,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-omnisciencehallucinationrate-2026-07-21","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":53.5,"normalizedScore":52.4729,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-122b-a10b-osworldverified-2026-07-21","modelSlug":"holo3-122b-a10b","modelName":"Holo3-122B-A10B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":78.85,"normalizedScore":86.6304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-tau2bench-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":4.1,"normalizedScore":4.1372,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aascicode-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":25.2,"normalizedScore":40.9781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-lcr-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8,"normalizedScore":10.568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-critpt-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-mmlupro-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":81.8,"normalizedScore":88.9015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aagpqadiamond-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":62.8,"normalizedScore":57.5881,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aahle-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.9,"normalizedScore":3.5857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aaomniscienceindex-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-62.3,"normalizedScore":19.5447,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-omniscienceaccuracy-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.4,"normalizedScore":12.3711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-omnisciencehallucinationrate-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81,"normalizedScore":19.3004,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aaifbench-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":33.5,"normalizedScore":25.2648,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aime2025-2026-07-21","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":85.3,"normalizedScore":79.3213,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-terminalbench2-2026-07-21","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":77.5,"normalizedScore":74.3772,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-claweval-2026-07-21","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":77.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-sweverified-2026-07-21","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":82.4,"normalizedScore":81.7803,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-swepro-2026-07-21","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":62.2,"normalizedScore":51.6043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-swemultilingual-2026-07-21","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.9,"normalizedScore":66.9154,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-nl2repo-2026-07-21","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":48.2,"normalizedScore":96.7742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-tau2bench-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74,"normalizedScore":74.672,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-gertlabs-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":65.59,"normalizedScore":84.4041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-researchclawbench-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":20.7,"normalizedScore":95.4023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-osworld2-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":13.9,"normalizedScore":18.5619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-vibecodebench-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":71.003,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-reactnativeevals-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":82.8,"normalizedScore":47.012,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aascicode-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50.1,"normalizedScore":82.968,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-frontiercode-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":38.5,"normalizedScore":48.6301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-lcr-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67,"normalizedScore":88.5073,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-critpt-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.1,"normalizedScore":15.7895,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aammmupro-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":76.4,"normalizedScore":86.3322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-designarenawebsite-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1325,"normalizedScore":89.934,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aagpqadiamond-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":88.5,"normalizedScore":92.4119,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aahle-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":31.2,"normalizedScore":55.9761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aaomniscienceindex-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.2,"normalizedScore":79.5918,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-omniscienceaccuracy-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.5,"normalizedScore":69.244,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-omnisciencehallucinationrate-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.9,"normalizedScore":54.4029,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aaifbench-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43.6,"normalizedScore":40.5446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-frontiermathv2tiers13-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":43.793,"normalizedScore":49.2056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-frontiermathv2tier4-2026-07-21","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":22.917,"normalizedScore":27.6108,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-browsecomp-2026-07-21","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":75.51,"normalizedScore":65.0837,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-hlewithtools-2026-07-21","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":47.6,"normalizedScore":51,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-vitabench-2026-07-21","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":38.75,"normalizedScore":71.7593,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-longbenchv2-2026-07-21","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.2,"normalizedScore":78.6802,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-hle-2026-07-21","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":47.6,"normalizedScore":70.2465,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-ifeval-2026-07-21","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.82,"normalizedScore":99.4681,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-terminalbench2-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":77.3,"normalizedScore":74.0214,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-osworldverified-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":64.7,"normalizedScore":55.8696,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-tau2bench-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86,"normalizedScore":86.781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-gertlabs-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":57.47,"normalizedScore":67.2443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-jobbench-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":33.7,"normalizedScore":54.5159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-sweverified-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":85,"normalizedScore":85.3964,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-swepro-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.8,"normalizedScore":37.1658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-swerebench-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.2,"normalizedScore":70.0422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-vibecodebench-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":61.767,"normalizedScore":86.9921,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aascicode-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.2,"normalizedScore":88.1956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-lcr-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-critpt-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":16.9,"normalizedScore":52.322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aammmupro-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.5,"normalizedScore":89.9654,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-designarenawebsite-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1193,"normalizedScore":68.1518,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aagpqadiamond-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":91.5,"normalizedScore":96.477,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aahle-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":39.9,"normalizedScore":73.3068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.9,"normalizedScore":76.2166,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.8,"normalizedScore":83.5052,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.9,"normalizedScore":12.1834,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aaifbench-2026-07-21","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.4,"normalizedScore":88.6536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-claweval-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":40.2,"normalizedScore":48.4637,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-vitabench-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":18.5,"normalizedScore":9.2593,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-tau2bench-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":78.9,"normalizedScore":79.6165,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-gertlabs-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":29.57,"normalizedScore":8.284,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-swerebench-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":60.9,"normalizedScore":81.4346,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-reactnativeevals-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71.5,"normalizedScore":1.992,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aascicode-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.7,"normalizedScore":63.7437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-lcr-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39,"normalizedScore":51.5192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-critpt-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-designarenawebsite-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1204,"normalizedScore":69.967,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aagpqadiamond-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":75.1,"normalizedScore":74.2547,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aahle-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":10.5,"normalizedScore":14.741,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aaomniscienceindex-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-46.7,"normalizedScore":31.7896,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-omniscienceaccuracy-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.2,"normalizedScore":36.0825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-omnisciencehallucinationrate-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.5,"normalizedScore":4.222,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aaifbench-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":49,"normalizedScore":48.7141,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-frontiermathv2tiers13-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":22.1,"normalizedScore":24.8315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-frontiermathv2tier4-2026-07-21","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.1,"normalizedScore":2.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-5-terminalbench2-2026-07-21","modelSlug":"composer-2-5","modelName":"Composer 2.5","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":69.3,"normalizedScore":59.7865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-5-swemultilingual-2026-07-21","modelSlug":"composer-2-5","modelName":"Composer 2.5","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":79.8,"normalizedScore":69.1542,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-tau3bench-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":91.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaagenticindex-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19,"normalizedScore":34.8595,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-tau2bench-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.2,"normalizedScore":95.0555,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-gdpvalaanormalized-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.4,"normalizedScore":34.2949,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-gdpvalaa-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":929,"normalizedScore":56.7125,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-gertlabs-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.1,"normalizedScore":28.4235,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aabriefcaseelo-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":506,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaenterpriseopsgym-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.7,"normalizedScore":32.0313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaharveylab-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.1,"normalizedScore":28.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aatau3banking-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-terminalbenchhard-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":20.2934,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-sweverified-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.6,"normalizedScore":75.1043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aacodingindex-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.9,"normalizedScore":55.952,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aascicode-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.6,"normalizedScore":65.2614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-lcr-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61,"normalizedScore":80.5812,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-critpt-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aammmupro-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":64.9,"normalizedScore":66.436,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aagpqadiamond-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":74.8,"normalizedScore":73.8482,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aahle-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":12.8,"normalizedScore":19.3227,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaomniscienceindex-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-36.3,"normalizedScore":39.9529,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-omniscienceaccuracy-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.1,"normalizedScore":37.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-omnisciencehallucinationrate-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82,"normalizedScore":18.0941,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaopennessindex-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaifbench-2026-07-21","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":68.8,"normalizedScore":78.6687,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-tau2bench-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":62.6,"normalizedScore":63.1685,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aacodingindex-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.72,"normalizedScore":45.5793,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aascicode-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.8,"normalizedScore":58.8533,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-lcr-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59.3,"normalizedScore":78.3355,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-critpt-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-mmlu-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":91.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-gpqa-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":75.7,"normalizedScore":71.7506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aagpqadiamond-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":74.7,"normalizedScore":73.7127,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aahle-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7.7,"normalizedScore":9.1633,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aaomniscienceindex-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.5,"normalizedScore":60.2041,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-omniscienceaccuracy-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.7,"normalizedScore":54.1237,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-omnisciencehallucinationrate-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":33.4138,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-ifeval-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":92.2,"normalizedScore":91.7258,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aaifbench-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.3,"normalizedScore":80.938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-frontiermathv2tiers13-2026-07-21","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":9.31,"normalizedScore":10.4607,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-claweval-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":63.8,"normalizedScore":81.4246,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-gdpvalaa-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1265,"normalizedScore":74.4715,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-tau3bench-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":72.9,"normalizedScore":28.2946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-terminalbench2-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":68.4,"normalizedScore":58.1851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaagenticindex-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.11,"normalizedScore":53.6758,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-tau2bench-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.2,"normalizedScore":95.0555,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-gdpvalaanormalized-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.3,"normalizedScore":61.3782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-apexagentsaa-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":2.4,"normalizedScore":3.6638,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-gertlabs-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":62.7,"normalizedScore":78.2967,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aabriefcaseelo-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":873,"normalizedScore":34.3633,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaitbench-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.2,"normalizedScore":64.4269,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-terminalbenchhard-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":43.2,"normalizedScore":44.4988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaterminalbench21-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.2,"normalizedScore":5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaharveylab-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.3,"normalizedScore":40.3361,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-swepro-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.2,"normalizedScore":38.2353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aacodingindex-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.19,"normalizedScore":75.1517,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aascicode-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50.2,"normalizedScore":83.1366,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-lcr-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.3,"normalizedScore":96.8296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-critpt-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4,"normalizedScore":12.3839,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-designarenawebsite-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1298,"normalizedScore":85.4785,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-hle-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":48,"normalizedScore":70.9507,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-hlenotools-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":34,"normalizedScore":53.5316,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aagpqadiamond-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86.6,"normalizedScore":89.8374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaomniscienceindex-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.6,"normalizedScore":71.2716,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-omniscienceaccuracy-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.6,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-omnisciencehallucinationrate-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.5,"normalizedScore":87.4548,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaopennessindex-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaifbench-2026-07-21","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":79.9,"normalizedScore":95.4614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-terminalbench2-2026-07-21","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":70.2,"normalizedScore":61.3879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-toolathlonverified-2026-07-21","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-toolathlonverified","benchmarkName":"Toolathlon-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":49.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-swemultilingual-2026-07-21","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.5,"normalizedScore":65.9204,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-swepro-2026-07-21","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":59.4,"normalizedScore":44.1176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-deepswe-2026-07-21","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":40.4,"normalizedScore":0,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-bfclv4-2026-07-21","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":39.22,"normalizedScore":33.7039,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-livecodebenchv6-2026-07-21","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":65.8,"normalizedScore":54.0885,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-gpqa-2026-07-21","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":71,"normalizedScore":65.0449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-gpqadiamond-2026-07-21","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":71,"normalizedScore":65.0449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-mmlupro-2026-07-21","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":74.2,"normalizedScore":78.0876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-ifeval-2026-07-21","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":85.58,"normalizedScore":72.1631,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-ifbench-2026-07-21","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":52.56,"normalizedScore":30.3863,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-aime2026-2026-07-21","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":89.1,"normalizedScore":82.8173,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-hmmtfeb2026-2026-07-21","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":71.6,"normalizedScore":64.2557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-imoanswerbench-2026-07-21","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":59.3,"normalizedScore":43.8757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-apex-2026-07-21","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":32.2,"normalizedScore":72.1088,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-jobbench-2026-07-21","modelSlug":"claude-4-1-opus","modelName":"Claude 4.1 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":21.9,"normalizedScore":28.9582,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-sweverified-2026-07-21","modelSlug":"claude-4-1-opus","modelName":"Claude 4.1 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.5,"normalizedScore":70.7928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-designarenawebsite-2026-07-21","modelSlug":"claude-4-1-opus","modelName":"Claude 4.1 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1207,"normalizedScore":70.462,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-terminalbench2-2026-07-21","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":61.7,"normalizedScore":46.2633,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-swemultilingual-2026-07-21","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.7,"normalizedScore":53.9801,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-swerebench-2026-07-21","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58,"normalizedScore":69.1983,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-reactnativeevals-2026-07-21","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":96.1,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-tau2bench-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":97.7,"normalizedScore":98.5873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gdpvalaanormalized-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.2,"normalizedScore":46.7949,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aaagenticindex-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.1,"normalizedScore":44.3514,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-apexagentsaa-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":17,"normalizedScore":35.1293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gdpvalaa-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1085,"normalizedScore":64.9577,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gertlabs-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":43.86,"normalizedScore":38.4827,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-researchclawbench-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":12.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-scicode-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":47.3,"normalizedScore":61.3293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aacodingindex-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.25,"normalizedScore":49.2343,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aascicode-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.3,"normalizedScore":78.2462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-lcr-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.3,"normalizedScore":84.9406,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-critpt-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":8,"normalizedScore":24.7678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-mmmupro-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.1,"normalizedScore":48.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-designarenawebsite-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1225,"normalizedScore":73.4323,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aammmupro-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.1,"normalizedScore":89.2734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gpqa-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.1,"normalizedScore":92.2956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-hle-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":35,"normalizedScore":48.0634,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-omniscienceaccuracy-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.6,"normalizedScore":53.9519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-omnisciencehallucinationrate-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25,"normalizedScore":86.8516,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aagpqadiamond-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":90.1,"normalizedScore":94.5799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aaomniscienceindex-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.3,"normalizedScore":82.81,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-ifbench-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":81.3,"normalizedScore":92.0601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aaifbench-2026-07-21","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":81.3,"normalizedScore":97.5794,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-terminalbench2-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":65.4,"normalizedScore":52.847,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-qwenclawbench-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59,"normalizedScore":57.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-qwenwebbench-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1532,"normalizedScore":78.9474,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-tau2bench-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-swepro-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.3,"normalizedScore":38.5027,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-scicode-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":47,"normalizedScore":60.423,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-nl2repo-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":42.9,"normalizedScore":72.3502,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aascicode-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.9,"normalizedScore":77.5717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-lcr-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-critpt-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.7,"normalizedScore":11.4551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-supergpqa-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":73.9,"normalizedScore":70.6374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aagpqadiamond-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":88.8,"normalizedScore":92.8184,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aahle-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.9,"normalizedScore":51.3944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aaomniscienceindex-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.2,"normalizedScore":76.4521,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-omniscienceaccuracy-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.7,"normalizedScore":59.2784,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-omnisciencehallucinationrate-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.2,"normalizedScore":63.6912,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aaifbench-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.6,"normalizedScore":90.469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-frontiermathv2tiers13-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":23.103,"normalizedScore":25.9584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-frontiermathv2tier4-2026-07-21","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-flash-aaagenticindex-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":11.99,"normalizedScore":21.8128,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-tau2bench-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83.9,"normalizedScore":84.662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-gdpvalaanormalized-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.7,"normalizedScore":26.7628,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-gdpvalaa-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":833,"normalizedScore":51.6385,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-sweverified-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.4,"normalizedScore":69.2629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aacodingindex-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.84,"normalizedScore":60.1994,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aascicode-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":25.9,"normalizedScore":42.1585,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-lcr-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.3,"normalizedScore":41.3474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-critpt-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-gpqa-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":83.7,"normalizedScore":83.1645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-mmlupro-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":84.9,"normalizedScore":93.3125,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aagpqadiamond-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":65.6,"normalizedScore":61.3821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aahle-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":8,"normalizedScore":9.761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aaomniscienceindex-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-48.5,"normalizedScore":30.3768,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-omniscienceaccuracy-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.2,"normalizedScore":20.6186,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-omnisciencehallucinationrate-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.1,"normalizedScore":26.4174,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aaifbench-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.9,"normalizedScore":34.947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aime2025-2026-07-21","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":94.1,"normalizedScore":94.8745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-interfaze-beta-spider2lite-2026-07-21","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-spider2lite","benchmarkName":"Spider 2.0-Lite","benchmarkCategory":"coding","benchmarkOrganisation":"Spider 2.0 authors","benchmarkVersion":"2024","score":52.9,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-ocrbenchv2-2026-07-21","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"ocrbench-v2","benchmarkName":"OCRBench v2","benchmarkCategory":"multimodal","benchmarkOrganisation":"OCRBench authors","benchmarkVersion":"2025","score":70.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-olmocr-2026-07-21","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-olmocr","benchmarkName":"olmOCR-Bench","benchmarkCategory":"multimodal","benchmarkOrganisation":"Allen Institute for AI","benchmarkVersion":"2025","score":85.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-refcocoavg-2026-07-21","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":82.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-voxpopuliwer-2026-07-21","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-voxpopuliwer","benchmarkName":"VoxPopuli-Cleaned-AA Word Error Rate","benchmarkCategory":"multimodal","benchmarkOrganisation":"Artificial Analysis / VoxPopuli dataset authors","benchmarkVersion":"2026","score":2.4,"normalizedScore":50,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-mmmupro-2026-07-21","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":71.1,"normalizedScore":26.129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-gpqa-2026-07-21","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":89.9,"normalizedScore":92.0103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-gpqadiamond-2026-07-21","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":89.9,"normalizedScore":92.0103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-mmmlu-2026-07-21","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-sobvalueacc-2026-07-21","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-sobvalueacc","benchmarkName":"Structured Output Benchmark Value Accuracy","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Interfaze","benchmarkVersion":"2026","score":79.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-pro-deep-think-arcagi2-2026-07-21","modelSlug":"gemini-3-pro-deep-think","modelName":"Gemini 3 Pro Deep Think","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":45.1,"normalizedScore":44.1176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-deep-think-critpt-2026-07-21","modelSlug":"gemini-3-pro-deep-think","modelName":"Gemini 3 Pro Deep Think","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":25.7,"normalizedScore":79.5666,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"excluded"},{"resultId":"benchlm-ref-muse-spark-terminalbench2-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59,"normalizedScore":41.4591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-tau2bench-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":91.5,"normalizedScore":92.331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-deepsearchqa-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":74.8,"normalizedScore":37.2671,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-cybergym-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":43.5,"normalizedScore":0.6865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-claweval-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":63.8,"normalizedScore":81.4246,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aaagenticindex-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.69,"normalizedScore":52.8941,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-gdpvalaanormalized-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.2,"normalizedScore":51.6026,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-gdpvalaa-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1144,"normalizedScore":68.0761,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-sweverified-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.4,"normalizedScore":74.8261,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-swepro-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.4,"normalizedScore":25.4011,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-livecodebenchpro-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":80,"normalizedScore":84.1456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-vibecodebench-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":19.674,"normalizedScore":27.7087,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aacodingindex-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.62,"normalizedScore":72.8836,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aascicode-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":51.5,"normalizedScore":85.3288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-arcagi2-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":42.5,"normalizedScore":40.4762,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-lcr-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-critpt-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":11.3,"normalizedScore":34.9845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-charxiv-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":86.4,"normalizedScore":82.598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-mmmupro-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":80.4,"normalizedScore":56.129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-erqa-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":64.7,"normalizedScore":71.978,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-simplevqa-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":71.3,"normalizedScore":59.375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-screenspotpro-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":84.1,"normalizedScore":90.9953,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-zerobench-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"2026","score":33,"normalizedScore":55.5556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-medxpertqamm-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":78.4,"normalizedScore":91.1043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aammmupro-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.5,"normalizedScore":93.4256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-gpqadiamond-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":89.5,"normalizedScore":91.4396,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-hle-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":50.4,"normalizedScore":75.1761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-hlenotools-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":42.8,"normalizedScore":69.8885,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-healthbenchhard-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":42.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-medxpertqatext-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":52.6,"normalizedScore":11.2676,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aaomniscienceindex-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.1,"normalizedScore":71.6641,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-omniscienceaccuracy-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.6,"normalizedScore":71.134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-omnisciencehallucinationrate-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.2,"normalizedScore":28.7093,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aaifbench-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.9,"normalizedScore":89.41,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-frontiermathv2tiers13-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":39,"normalizedScore":43.8202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-frontiermathv2tier4-2026-07-21","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.6,"normalizedScore":17.5904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-tau2bench-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":52.3,"normalizedScore":52.775,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-gertlabs-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.66,"normalizedScore":29.6069,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-jobbench-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":18.4,"normalizedScore":21.3775,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-sweverified-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":72.7,"normalizedScore":68.2893,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aascicode-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.3,"normalizedScore":61.3828,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-lcr-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.3,"normalizedScore":58.5205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-critpt-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aammmupro-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":62.4,"normalizedScore":62.1107,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-designarenawebsite-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1175,"normalizedScore":65.1815,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aagpqadiamond-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68.3,"normalizedScore":65.0407,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aahle-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4,"normalizedScore":1.7928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aaomniscienceindex-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-9.2,"normalizedScore":61.2245,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-omniscienceaccuracy-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.4,"normalizedScore":32.9897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-omnisciencehallucinationrate-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.8,"normalizedScore":67.7925,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aaifbench-2026-07-21","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":45.4,"normalizedScore":43.2678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-bigcodebench-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"bigcodebench","benchmarkName":"BigCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"BigCodeBench authors","benchmarkVersion":"2026","score":59.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-humaneval-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-humaneval","benchmarkName":"Evaluating Large Language Models Trained on Code","benchmarkCategory":"coding","benchmarkOrganisation":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","benchmarkVersion":"2021","score":76.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-bbh-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":87.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-drop-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-drop","benchmarkName":"Discrete Reasoning Over Paragraphs","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-hellaswag-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hellaswag","benchmarkName":"HellaSwag","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-winogrande-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-winogrande","benchmarkName":"WinoGrande","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":81.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-cluewsc-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cluewsc","benchmarkName":"CLUEWSC","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-longbenchv2-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":51.5,"normalizedScore":34.5178,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-agieval-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-agieval","benchmarkName":"AGIEval","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":83.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmlu-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":90.1,"normalizedScore":85.4701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmluredux-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":90.8,"normalizedScore":78.1462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmlupro-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":73.5,"normalizedScore":77.0916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmmlu-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90.3,"normalizedScore":92,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-ceval-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":93.1,"normalizedScore":93.9394,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-cmmlu-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmmlu","benchmarkName":"Chinese Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-multiloko-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-multiloko","benchmarkName":"MultiLoKo","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":51.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-simpleqa-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":55.2,"normalizedScore":92.2414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-supergpqa-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":53.9,"normalizedScore":42.8055,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-factsparametric-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-factsparametric","benchmarkName":"FACTS Parametric","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":62.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-triviaqa-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-triviaqa","benchmarkName":"TriviaQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mgsm-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mgsm","benchmarkName":"Multilingual Grade School Math","benchmarkCategory":"knowledge","benchmarkOrganisation":"Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","benchmarkVersion":"2022","score":84.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-gsm8k-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gsm8k","benchmarkName":"Grade School Math 8K","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":92.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mathbenchmark-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mathbenchmark","benchmarkName":"MATH","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":64.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-cmath-2026-07-21","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmath","benchmarkName":"CMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-claweval-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":49.2,"normalizedScore":61.0335,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-tau2bench-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":43.3,"normalizedScore":43.6932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-gertlabs-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":56.63,"normalizedScore":65.4691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-jobbench-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":11.4,"normalizedScore":6.2162,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-vibecodebench-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":20.204,"normalizedScore":28.4551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aascicode-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49.9,"normalizedScore":82.6307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-lcr-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48,"normalizedScore":63.4082,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-critpt-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aammmupro-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.6,"normalizedScore":90.1384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-designarenawebsite-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1226,"normalizedScore":73.5974,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aagpqadiamond-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81.2,"normalizedScore":82.5203,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aahle-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.1,"normalizedScore":21.9124,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aaomniscienceindex-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-3.6,"normalizedScore":65.6201,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-omniscienceaccuracy-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.5,"normalizedScore":72.6804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.2,"normalizedScore":8.2027,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aaifbench-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":55.1,"normalizedScore":57.9425,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-frontiermathv2tiers13-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":35.64,"normalizedScore":40.0449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-frontiermathv2tier4-2026-07-21","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-claweval-2026-07-21","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.3,"normalizedScore":79.3296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-mmclawbench-2026-07-21","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-mmclawbench","benchmarkName":"MM-ClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":23.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-terminalbench2-2026-07-21","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":65.8,"normalizedScore":53.5587,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-gertlabs-2026-07-21","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":46.89,"normalizedScore":44.8859,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-researchclawbench-2026-07-21","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":16.9,"normalizedScore":51.7241,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-swepro-2026-07-21","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.1,"normalizedScore":35.2941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-videommewithsub-2026-07-21","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.7,"normalizedScore":88.4615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-charxiv-2026-07-21","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":81,"normalizedScore":69.3627,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-mmmupro-2026-07-21","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":77.9,"normalizedScore":48.0645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-designarenawebsite-2026-07-21","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1291,"normalizedScore":84.3234,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aaagenticindex-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.01,"normalizedScore":38.6004,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-tau2bench-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":81.9,"normalizedScore":82.6438,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-gertlabs-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":41.24,"normalizedScore":32.9459,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.4,"normalizedScore":39.1026,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-gdpvalaa-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":987,"normalizedScore":59.778,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-vibecodebench-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":24.606,"normalizedScore":34.6549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aacodingindex-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.39,"normalizedScore":59.5493,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aascicode-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.3,"normalizedScore":71.5008,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-lcr-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75,"normalizedScore":99.0753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-critpt-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.9,"normalizedScore":15.1703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aammmupro-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.5,"normalizedScore":84.7751,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-designarenawebsite-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1217,"normalizedScore":72.1122,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aagpqadiamond-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87.3,"normalizedScore":90.7859,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aahle-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":26.5,"normalizedScore":46.6135,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.6,"normalizedScore":72.8414,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.6,"normalizedScore":59.1065,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.3,"normalizedScore":55.1267,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aaifbench-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.9,"normalizedScore":84.8714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":31.034,"normalizedScore":34.8697,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-frontiermathv2tier4-2026-07-21","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":12.5,"normalizedScore":15.0602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-terminalbench2-2026-07-21","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":64.2,"normalizedScore":50.7117,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-claweval-2026-07-21","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":69.8,"normalizedScore":89.8045,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-sweverified-2026-07-21","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":75.6,"normalizedScore":72.3227,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-swepro-2026-07-21","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":50.4,"normalizedScore":20.0535,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-swemultilingual-2026-07-21","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":69.3,"normalizedScore":43.0348,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-nl2repo-2026-07-21","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":34.6,"normalizedScore":34.1014,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-jobbench-2026-07-21","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":16,"normalizedScore":16.1793,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-sweverified-2026-07-21","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.3,"normalizedScore":69.1238,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-designarenawebsite-2026-07-21","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1152,"normalizedScore":61.3861,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-frontiermathv2tiers13-2026-07-21","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":5.903,"normalizedScore":6.6326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-frontiermathv2tier4-2026-07-21","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aaagenticindex-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.17,"normalizedScore":10.9808,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-apexagentsaa-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":12.2,"normalizedScore":24.7845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-tau2bench-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":31.3,"normalizedScore":31.5843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-gdpvalaanormalized-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.1,"normalizedScore":11.3782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-gdpvalaa-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":642,"normalizedScore":41.5433,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-gertlabs-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":38.46,"normalizedScore":27.071,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-vibecodebench-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aacodingindex-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.69,"normalizedScore":38.3126,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aascicode-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":41.9,"normalizedScore":69.14,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-lcr-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.3,"normalizedScore":86.2616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-critpt-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-charxiv-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":73.2,"normalizedScore":50.2451,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aammmupro-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.5,"normalizedScore":84.7751,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aagpqadiamond-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":82.2,"normalizedScore":83.8753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aahle-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":16.2,"normalizedScore":26.0956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aaomniscienceindex-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-15.5,"normalizedScore":56.2794,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-omniscienceaccuracy-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":36.4,"normalizedScore":57.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.6,"normalizedScore":18.5766,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aaifbench-2026-07-21","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":77.2,"normalizedScore":91.3767,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-osworld-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":"2026","score":47.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-tau2bench-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":45.3,"normalizedScore":45.7114,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaanormalized-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaa-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":467,"normalizedScore":32.2939,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-scicode-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":32,"normalizedScore":15.1057,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aascicode-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":27.8,"normalizedScore":45.3626,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aacodingindex-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.75,"normalizedScore":8.0613,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-lcr-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.7,"normalizedScore":47.1598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-critpt-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmmu-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":70.8,"normalizedScore":71.4982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlongbenchdoc-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmlongbenchdoc","benchmarkName":"MMLongBench-Doc","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":57.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-charxiv-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":76.25,"normalizedScore":57.7206,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-screenspotpro-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":57.8,"normalizedScore":28.673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-videommenosub-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-videommenosub","benchmarkName":"Video-MME without subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":72.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-ai2dtest-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-ai2dtest","benchmarkName":"AI2D test split","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-refcocoavg-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":90.5,"normalizedScore":80.7692,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aammmupro-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":53.2,"normalizedScore":46.1938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlupro-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":77.3,"normalizedScore":82.4986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqa-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":72.2,"normalizedScore":66.757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqadiamond-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":72.2,"normalizedScore":66.757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aahle-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.3,"normalizedScore":4.3825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaomniscienceindex-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-56,"normalizedScore":24.4898,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-omniscienceaccuracy-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.8,"normalizedScore":19.9313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-omnisciencehallucinationrate-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.1,"normalizedScore":16.7672,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-ifbench-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":74.2,"normalizedScore":76.824,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaifbench-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":63.2,"normalizedScore":70.1967,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aime2025-2026-07-21","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":82.1,"normalizedScore":73.6656,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-terminalbench2-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59.5,"normalizedScore":42.3488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-browsecomp-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":75.82,"normalizedScore":65.7322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-deepsearchqa-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":92.82,"normalizedScore":93.2298,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-gdpvalaanormalized-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.9,"normalizedScore":41.5064,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-toolathlon-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":49.5,"normalizedScore":46.4066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-claweval-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":67.1,"normalizedScore":86.0335,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-hlewithtools-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":47.2,"normalizedScore":49,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-gertlabs-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":51.57,"normalizedScore":54.776,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aaagenticindex-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.53,"normalizedScore":39.5682,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-tau2bench-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-gdpvalaa-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1017,"normalizedScore":61.3636,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-apexagentsaa-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":14.8,"normalizedScore":30.3879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-swepro-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.3,"normalizedScore":35.8289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aacodingindex-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.57,"normalizedScore":45.3626,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aascicode-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40,"normalizedScore":65.9359,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-lcr-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.7,"normalizedScore":84.148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-critpt-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.3,"normalizedScore":7.1207,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-simplevqa-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":79.2,"normalizedScore":90.2344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-vstar-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":95.3,"normalizedScore":94.6488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aammmupro-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.3,"normalizedScore":84.4291,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-designarenawebsite-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1211,"normalizedScore":71.1221,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aagpqadiamond-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":80.9,"normalizedScore":82.1138,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aahle-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":19.9,"normalizedScore":33.4661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aaomniscienceindex-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-37.5,"normalizedScore":39.011,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-omniscienceaccuracy-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.4,"normalizedScore":38.1443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-omnisciencehallucinationrate-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.4,"normalizedScore":15.199,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aaifbench-2026-07-21","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":67.3,"normalizedScore":76.3994,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-gpqa-2026-07-21","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":77.5,"normalizedScore":74.3187,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-supergpqa-2026-07-21","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":62.6,"normalizedScore":54.9123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-mmlupro-2026-07-21","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":83,"normalizedScore":90.609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-mmluprox-2026-07-21","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":79.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o4-mini-high-frontiermathv2tiers13-2026-07-21","modelSlug":"o4-mini-high","modelName":"o4-mini (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":24.828,"normalizedScore":27.8966,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o4-mini-high-frontiermathv2tier4-2026-07-21","modelSlug":"o4-mini-high","modelName":"o4-mini (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":6.25,"normalizedScore":7.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-terminalbench2-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":54,"normalizedScore":32.5623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-osworldverified-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":74,"normalizedScore":76.087,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-gdpvalaa-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1140,"normalizedScore":67.8647,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aaagenticindex-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.82,"normalizedScore":49.4137,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-gdpvalaanormalized-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32,"normalizedScore":51.2821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aabriefcaseelo-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":634,"normalizedScore":11.985,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aatau3banking-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.5,"normalizedScore":11.0526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-swepro-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":54.2,"normalizedScore":30.2139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aacodingindex-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.32,"normalizedScore":59.4481,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aascicode-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.9,"normalizedScore":67.4536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-mrcrv2-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":72.2,"normalizedScore":57.3705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-lcr-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62,"normalizedScore":81.9022,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-critpt-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aammmupro-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":79,"normalizedScore":90.8304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aagpqadiamond-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":83.8,"normalizedScore":86.0434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aahle-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":17.5,"normalizedScore":28.6853,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aaomniscienceindex-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.9,"normalizedScore":73.8619,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-omniscienceaccuracy-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.3,"normalizedScore":46.5636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.5,"normalizedScore":76.5983,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-bfclv4-2026-07-21","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":45.6,"normalizedScore":45.5253,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-mmluredux-2026-07-21","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":60.8139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-gpqa-2026-07-21","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":57.6,"normalizedScore":45.9267,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-gpqadiamond-2026-07-21","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":57.6,"normalizedScore":45.9267,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-ifeval-2026-07-21","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":76.5,"normalizedScore":45.331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-terminalbench2-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":57,"normalizedScore":37.9004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-tau2bench-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-toolathlon-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":46.3,"normalizedScore":39.8357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-mlebenchlite-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mlebenchlite","benchmarkName":"MLE-Bench Lite","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":66.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-mmclawbench-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mmclawbench","benchmarkName":"MM-ClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":62.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-claweval-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":48.7,"normalizedScore":60.3352,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aaagenticindex-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.58,"normalizedScore":47.1059,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-apexagentsaa-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":10.6,"normalizedScore":21.3362,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gdpvalaanormalized-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.9,"normalizedScore":52.7244,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gdpvalaa-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1158,"normalizedScore":68.8161,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gertlabs-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":40.4,"normalizedScore":31.1708,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-sweverifiedarcee-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75.4,"normalizedScore":98.3871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-swepro-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.2,"normalizedScore":35.5615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-swerebench-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":51.9,"normalizedScore":43.4599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-swemultilingual-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":76.5,"normalizedScore":60.9453,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-multiswebench-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"multi-swe-bench","benchmarkName":"Multi-SWE-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"Multi-SWE-Bench","benchmarkVersion":"2026","score":52.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-vibepro-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibepro","benchmarkName":"VIBE-Pro","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":55.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-nl2repo-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":39.8,"normalizedScore":58.0645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-vibecodebench-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":27.037,"normalizedScore":38.0787,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-reactnativeevals-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71.4,"normalizedScore":1.5936,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aacodingindex-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.62,"normalizedScore":64.2155,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aascicode-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47,"normalizedScore":77.7403,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-lcr-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.7,"normalizedScore":90.753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-critpt-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-designarenawebsite-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1275,"normalizedScore":81.6832,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gpqadiamond-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-mmluproarcee-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":80.8,"normalizedScore":40.2878,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aahle-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.1,"normalizedScore":49.8008,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aaomniscienceindex-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.7,"normalizedScore":68.9953,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-omniscienceaccuracy-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.1,"normalizedScore":39.3471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-omnisciencehallucinationrate-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.4,"normalizedScore":75.5127,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aaifbench-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.7,"normalizedScore":89.1074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aime2025arcee-2026-07-21","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":80,"normalizedScore":73.8786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-bfclv4-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":49.73,"normalizedScore":53.1777,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-tau2bench-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":16.1,"normalizedScore":16.2462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aascicode-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":7.8,"normalizedScore":11.6358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-lcr-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-critpt-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aagpqadiamond-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":51.3,"normalizedScore":42.0054,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aahle-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.9,"normalizedScore":7.5697,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aaomniscienceindex-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-33.3,"normalizedScore":42.3077,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-omniscienceaccuracy-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.4,"normalizedScore":10.6529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-omnisciencehallucinationrate-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47,"normalizedScore":60.3136,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-ifeval-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":91.84,"normalizedScore":90.6619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-ifbench-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":56.47,"normalizedScore":38.7768,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aaifbench-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":55.6,"normalizedScore":58.6989,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-math500-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-math500","benchmarkName":"MATH-500 Problem Set","benchmarkCategory":"mathematics","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2021","score":88.76,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aime2025-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":42.53,"normalizedScore":3.7292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aime2026-2026-07-21","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":50,"normalizedScore":16.2981,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-jobbench-2026-07-21","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":18.5,"normalizedScore":21.5941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-vibecodebench-2026-07-21","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":15.738,"normalizedScore":22.1653,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-frontiermathv2tiers13-2026-07-21","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":21.034,"normalizedScore":23.6337,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-frontiermathv2tier4-2026-07-21","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-tau2bench-2026-07-21","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":28.7,"normalizedScore":28.9606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-sweverified-2026-07-21","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":49.3,"normalizedScore":35.7441,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aascicode-2026-07-21","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.9,"normalizedScore":65.7673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-mmlu-2026-07-21","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":86.9,"normalizedScore":58.1197,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-gpqa-2026-07-21","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":77.2,"normalizedScore":73.8907,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aagpqadiamond-2026-07-21","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":74.8,"normalizedScore":73.8482,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aahle-2026-07-21","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":8.7,"normalizedScore":11.1554,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-ifeval-2026-07-21","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":93.9,"normalizedScore":96.7494,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aime2024-2026-07-21","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aime2024","benchmarkName":"American Invitational Mathematics Examination 2024","benchmarkCategory":"mathematics","benchmarkOrganisation":"Mathematical Association of America","benchmarkVersion":"2024","score":87.3,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-tau2bench-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":61.1,"normalizedScore":61.6549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aascicode-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":34.5,"normalizedScore":56.661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-lcr-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51,"normalizedScore":67.3712,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-critpt-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-designarenawebsite-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1080,"normalizedScore":49.505,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aagpqadiamond-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.6,"normalizedScore":76.2873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aahle-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7,"normalizedScore":7.7689,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aaomniscienceindex-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-27.5,"normalizedScore":46.8603,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-omniscienceaccuracy-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":40.5498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-omnisciencehallucinationrate-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.2,"normalizedScore":27.503,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aaifbench-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":41.5,"normalizedScore":37.3676,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-frontiermathv2tiers13-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":21.404,"normalizedScore":24.0494,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-frontiermathv2tier4-2026-07-21","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-humaneval-2026-07-21","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-humaneval","benchmarkName":"Evaluating Large Language Models Trained on Code","benchmarkCategory":"coding","benchmarkOrganisation":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","benchmarkVersion":"2021","score":73.8,"normalizedScore":58.9041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-bbh-2026-07-21","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":78.8,"normalizedScore":74.7826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-drop-2026-07-21","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-drop","benchmarkName":"Discrete Reasoning Over Paragraphs","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":66.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-gpqa-2026-07-21","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":43.4,"normalizedScore":25.667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-gpqadiamond-2026-07-21","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":43.4,"normalizedScore":25.667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-mmlupro-2026-07-21","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":51.4,"normalizedScore":45.646,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-agieval-2026-07-21","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-agieval","benchmarkName":"AGIEval","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":66.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-gsm8k-2026-07-21","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-gsm8k","benchmarkName":"Grade School Math 8K","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":86.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-tau2bench-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74.9,"normalizedScore":75.5802,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-gertlabs-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":42.34,"normalizedScore":35.2705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-reactnativeevals-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":72.6,"normalizedScore":6.3745,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aascicode-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.7,"normalizedScore":75.5481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-lcr-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68,"normalizedScore":89.8283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-critpt-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2,"normalizedScore":6.192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aammmupro-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":68.8,"normalizedScore":73.1834,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aagpqadiamond-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87.7,"normalizedScore":91.3279,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aahle-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.9,"normalizedScore":41.4343,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aaomniscienceindex-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.8,"normalizedScore":71.4286,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-omniscienceaccuracy-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.4,"normalizedScore":65.6357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-omnisciencehallucinationrate-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.2,"normalizedScore":39.5657,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aaifbench-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":53.7,"normalizedScore":55.8245,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-frontiermathv2tiers13-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":19.655,"normalizedScore":22.0843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-frontiermathv2tier4-2026-07-21","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-tau2bench-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":80.7,"normalizedScore":81.4329,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aascicode-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":41,"normalizedScore":67.6223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-lcr-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":91.5456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-critpt-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aammmupro-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":70.1,"normalizedScore":75.4325,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-designarenawebsite-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1067,"normalizedScore":47.3597,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aagpqadiamond-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":82.7,"normalizedScore":84.5528,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aahle-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":20,"normalizedScore":33.6653,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aaomniscienceindex-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-15.3,"normalizedScore":56.4364,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-omniscienceaccuracy-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.4,"normalizedScore":60.4811,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-omnisciencehallucinationrate-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.1,"normalizedScore":11.9421,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aaifbench-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":71.4,"normalizedScore":82.6021,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aamath500-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aamath500","benchmarkName":"Artificial Analysis MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99.2,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-frontiermathv2tiers13-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":18.685,"normalizedScore":20.9944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-frontiermathv2tier4-2026-07-21","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aaagenticindex-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.97,"normalizedScore":19.9144,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-tau2bench-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":43.6,"normalizedScore":43.996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-gdpvalaanormalized-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.1,"normalizedScore":20.9936,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-gdpvalaa-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":761,"normalizedScore":47.833,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aacodingindex-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.32,"normalizedScore":45.0014,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aascicode-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40,"normalizedScore":65.9359,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-lcr-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.7,"normalizedScore":73.5799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-critpt-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-mmmupro-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":73.8,"normalizedScore":34.8387,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aammmupro-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":69.2,"normalizedScore":73.8754,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-mmlupro-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":82.6,"normalizedScore":90.0398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-hle-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":17.2,"normalizedScore":16.7254,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-hlenotools-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":8.7,"normalizedScore":6.5056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aagpqadiamond-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":79.2,"normalizedScore":79.8103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aaomniscienceindex-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-48.1,"normalizedScore":30.6907,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-omniscienceaccuracy-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.2,"normalizedScore":25.7732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.9,"normalizedScore":19.421,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aaifbench-2026-07-21","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.4,"normalizedScore":84.115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-terminalbench2-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":54.4,"normalizedScore":33.274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gertlabs-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":36.91,"normalizedScore":23.7954,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aaagenticindex-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.73,"normalizedScore":56.6909,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gdpvalaanormalized-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.7,"normalizedScore":57.2115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gdpvalaa-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1214,"normalizedScore":71.7759,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-sweverified-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.4,"normalizedScore":70.6537,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-scicode-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":41.2,"normalizedScore":42.9003,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aascicode-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.6,"normalizedScore":78.7521,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aacodingindex-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.8,"normalizedScore":73.1436,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-lcr-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-critpt-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.9,"normalizedScore":15.1703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gpqa-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.2,"normalizedScore":88.1581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gpqadiamond-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87.2,"normalizedScore":88.1581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-hle-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":25.5,"normalizedScore":31.338,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-omniscienceaccuracy-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.5,"normalizedScore":48.6254,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-omnisciencehallucinationrate-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73,"normalizedScore":28.9505,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aaomniscienceindex-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-18.5,"normalizedScore":53.9246,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-ifbench-2026-07-21","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":63.1,"normalizedScore":53.0043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aaagenticindex-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.17,"normalizedScore":1.675,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-tau2bench-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":17.3,"normalizedScore":17.4571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-gdpvalaa-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":41,"normalizedScore":9.778,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aacodingindex-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":11.14,"normalizedScore":4.2907,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aascicode-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":25.9,"normalizedScore":42.1585,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-lcr-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17,"normalizedScore":22.4571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-critpt-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aammmupro-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":40.1,"normalizedScore":23.5294,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-designarenawebsite-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1003,"normalizedScore":36.7987,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-mmlu-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":80.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-gpqa-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":50.3,"normalizedScore":35.5115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aagpqadiamond-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":51.2,"normalizedScore":41.8699,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aahle-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.9,"normalizedScore":1.5936,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aaomniscienceindex-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-56.4,"normalizedScore":24.1758,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.3,"normalizedScore":17.354,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.4,"normalizedScore":20.0241,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-ifeval-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":83.2,"normalizedScore":65.13,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aaifbench-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":32,"normalizedScore":22.9955,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":1.034,"normalizedScore":1.1618,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-terminalbench2-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":50,"normalizedScore":25.4448,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-osworldverified-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":61.4,"normalizedScore":48.6957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-vitabench-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":17,"normalizedScore":4.6296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-gertlabs-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":48.51,"normalizedScore":48.3094,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-jobbench-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":27.7,"normalizedScore":41.5205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-sweverified-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.2,"normalizedScore":74.548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-arcagi2-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":13.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-designarenawebsite-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1219,"normalizedScore":72.4422,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-gpqa-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":83.4,"normalizedScore":82.7365,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-aime2025-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":87,"normalizedScore":82.3259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-frontiermathv2tiers13-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":13.495,"normalizedScore":15.1629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-frontiermathv2tier4-2026-07-21","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-terminalbench2-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":47.1,"normalizedScore":20.2847,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-deepsearchqa-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":62.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-gertlabs-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":38.36,"normalizedScore":26.8597,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-livecodebenchpro-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":74.2,"normalizedScore":75.6312,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-sweverified-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":76.7,"normalizedScore":73.8526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-swepro-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":51.8,"normalizedScore":23.7968,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-vibecodebench-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":4.064,"normalizedScore":5.7237,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-arcagi2-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":53.3,"normalizedScore":55.6022,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-mmmupro-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":75.2,"normalizedScore":39.3548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-charxiv-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":60.9,"normalizedScore":20.098,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-erqa-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":54.1,"normalizedScore":13.7363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-simplevqa-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":57.4,"normalizedScore":5.0781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-medxpertqamm-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":65.8,"normalizedScore":52.454,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-designarenawebsite-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1257,"normalizedScore":78.7129,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-gpqadiamond-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":88.5,"normalizedScore":90.0128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-hlenotools-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":31.6,"normalizedScore":49.0706,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-healthbenchhard-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":20.3,"normalizedScore":19.6429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-medxpertqatext-2026-07-21","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":50.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-tau2airline-2026-07-21","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-tau2airline","benchmarkName":"τ²-Bench Airline Domain","benchmarkCategory":"agents","benchmarkOrganisation":"Victor Barres, Honghua Dong, Soham Ray, Xujie Si, Karthik Narasimhan","benchmarkVersion":"2025","score":56.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-livecodebenchv6-2026-07-21","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":65.7,"normalizedScore":53.9209,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-sweverified-2026-07-21","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":53.2,"normalizedScore":41.1683,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-mmlupro-2026-07-21","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":68.1,"normalizedScore":69.4081,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-gpqa-2026-07-21","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":57.3,"normalizedScore":45.4986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-gpqadiamond-2026-07-21","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":57.3,"normalizedScore":45.4986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-aime2026-2026-07-21","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":76.4,"normalizedScore":61.2113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaagenticindex-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.45,"normalizedScore":26.3912,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-tau2bench-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":59.9,"normalizedScore":60.444,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gdpvalaanormalized-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.2,"normalizedScore":24.359,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gdpvalaa-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":804,"normalizedScore":50.1057,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gertlabs-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":35.26,"normalizedScore":20.3085,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaenterpriseopsgym-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.3,"normalizedScore":10.9375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaitbench-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.3,"normalizedScore":62.6482,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aatau3banking-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.1,"normalizedScore":3.6842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-terminalbenchhard-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":36.4,"normalizedScore":27.8729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-swerebench-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":41.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-reactnativeevals-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":75.2,"normalizedScore":16.7331,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aacodingindex-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.43,"normalizedScore":50.939,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aascicode-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.4,"normalizedScore":71.6695,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-lcr-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62,"normalizedScore":81.9022,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-critpt-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-mmmupro-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":76.9,"normalizedScore":44.8387,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aammmupro-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":73.4,"normalizedScore":81.1419,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gpqa-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":84.3,"normalizedScore":84.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-mmlupro-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.2,"normalizedScore":93.7393,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-hle-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":26.5,"normalizedScore":33.0986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-hlenotools-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":19.5,"normalizedScore":26.5799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aagpqadiamond-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.7,"normalizedScore":88.6179,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaomniscienceindex-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-45.4,"normalizedScore":32.81,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-omniscienceaccuracy-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.9,"normalizedScore":28.6942,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.6,"normalizedScore":18.5766,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaopennessindex-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaifbench-2026-07-21","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.6,"normalizedScore":88.9561,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-tau2bench-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":47.1,"normalizedScore":47.5277,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-gertlabs-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":25.65,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-sweverified-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":54.6,"normalizedScore":43.1154,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aascicode-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.1,"normalizedScore":62.7319,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-lcr-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61,"normalizedScore":80.5812,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-critpt-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aammmupro-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":61.2,"normalizedScore":60.0346,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-designarenawebsite-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1068,"normalizedScore":47.5248,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mmlu-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":90.2,"normalizedScore":86.3248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-gpqa-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":66.3,"normalizedScore":58.3393,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aagpqadiamond-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":66.6,"normalizedScore":62.7371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aahle-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aaomniscienceindex-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-36.2,"normalizedScore":40.0314,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.2,"normalizedScore":36.0825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":79.6,"normalizedScore":20.9891,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-ifeval-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":87.4,"normalizedScore":77.5414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aaifbench-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43,"normalizedScore":39.6369,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":5.517,"normalizedScore":6.1989,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-frontiermathv2tier4-2026-07-21","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tiers13-2026-07-21","modelSlug":"qwen3-235b-2507-reasoning","modelName":"Qwen3 235B 2507 (Reasoning)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":8.481,"normalizedScore":9.5292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tier4-2026-07-21","modelSlug":"qwen3-235b-2507-reasoning","modelName":"Qwen3 235B 2507 (Reasoning)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-tau2bench-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":14.9,"normalizedScore":15.0353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aascicode-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.1,"normalizedScore":47.5548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-lcr-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.9,"normalizedScore":60.6341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-critpt-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aammmupro-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":65.5,"normalizedScore":67.474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-designarenawebsite-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1145,"normalizedScore":60.231,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aagpqadiamond-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68.3,"normalizedScore":65.0407,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aahle-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.1,"normalizedScore":3.9841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aaomniscienceindex-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-42,"normalizedScore":35.4788,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-omniscienceaccuracy-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.5,"normalizedScore":40.0344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.3,"normalizedScore":4.4632,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aaifbench-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39,"normalizedScore":33.5855,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-frontiermathv2tiers13-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.844,"normalizedScore":5.4427,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-frontiermathv2tier4-2026-07-21","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aaagenticindex-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.72,"normalizedScore":2.6987,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-tau2bench-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":52.9,"normalizedScore":53.3804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.1,"normalizedScore":0.1603,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-gdpvalaa-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":503,"normalizedScore":34.1966,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-sweverified-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":23.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aacodingindex-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.21,"normalizedScore":17.3938,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aascicode-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.4,"normalizedScore":66.6105,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-lcr-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.3,"normalizedScore":55.8785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-critpt-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aammmupro-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":58.7,"normalizedScore":55.7093,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-designarenawebsite-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1027,"normalizedScore":40.7591,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-mmlu-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":87.5,"normalizedScore":63.2479,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-gpqa-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":64.2,"normalizedScore":55.3431,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aagpqadiamond-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":66.4,"normalizedScore":62.4661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aahle-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aaomniscienceindex-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-50.1,"normalizedScore":29.1209,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.5,"normalizedScore":24.5704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82,"normalizedScore":18.0941,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-ifeval-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":88.5,"normalizedScore":80.792,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aaifbench-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":38.3,"normalizedScore":32.5265,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.483,"normalizedScore":5.0371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-tau2bench-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":36.3,"normalizedScore":36.6297,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aascicode-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.2,"normalizedScore":62.9005,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-bbh-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":53,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mrcrv2-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":43.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-lcr-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.3,"normalizedScore":73.0515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-critpt-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mmmupro-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":69.1,"normalizedScore":19.6774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mathvision-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":79.7,"normalizedScore":27,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-medxpertqamm-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":48.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aammmupro-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":69.7,"normalizedScore":74.7405,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-gpqa-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":78.8,"normalizedScore":76.1735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-gpqadiamond-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":78.8,"normalizedScore":76.1735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mmlupro-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":77.2,"normalizedScore":82.3563,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-hlenotools-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":5.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mmmlu-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aahle-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.8,"normalizedScore":23.3068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aaomniscienceindex-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-51.9,"normalizedScore":27.708,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-omniscienceaccuracy-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16,"normalizedScore":21.9931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.8,"normalizedScore":19.5416,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aaifbench-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.5,"normalizedScore":85.7791,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aime2026-2026-07-21","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":77.5,"normalizedScore":63.0827,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-flash-frontiermathv2tiers13-2026-07-21","modelSlug":"qwen3-5-flash","modelName":"Qwen3.5 Flash","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":6.207,"normalizedScore":6.9742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-flash-frontiermathv2tier4-2026-07-21","modelSlug":"qwen3-5-flash","modelName":"Qwen3.5 Flash","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-terminalbench2-2026-07-21","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":45.8,"normalizedScore":17.9715,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-sweverified-2026-07-21","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.6,"normalizedScore":70.9318,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-swemultilingual-2026-07-21","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":63.1,"normalizedScore":27.6119,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-swepro-2026-07-21","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":49.2,"normalizedScore":16.8449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-tau2bench-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":76.9,"normalizedScore":77.5984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-vibecodebench-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":3.09,"normalizedScore":4.3519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aascicode-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.1,"normalizedScore":54.3002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-lcr-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.3,"normalizedScore":34.7424,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-critpt-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aagpqadiamond-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":63.2,"normalizedScore":58.1301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aahle-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.2,"normalizedScore":4.1833,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aaomniscienceindex-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-31.6,"normalizedScore":43.6421,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-omniscienceaccuracy-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.8,"normalizedScore":30.2405,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-omnisciencehallucinationrate-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.1,"normalizedScore":37.2738,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aaifbench-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.7,"normalizedScore":30.1059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-frontiermathv2tiers13-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":3.819,"normalizedScore":4.291,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-frontiermathv2tier4-2026-07-21","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.128,"normalizedScore":2.5639,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-3-beta-frontiermathv2tiers13-2026-07-21","modelSlug":"grok-3-beta","modelName":"Grok 3 [Beta]","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":3.793,"normalizedScore":4.2618,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-3-beta-frontiermathv2tier4-2026-07-21","modelSlug":"grok-3-beta","modelName":"Grok 3 [Beta]","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-bfclv4-2026-07-21","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":44.2,"normalizedScore":42.9313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-mmluredux-2026-07-21","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":78.1,"normalizedScore":30.2939,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-gpqa-2026-07-21","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":40.9,"normalizedScore":22.1002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-gpqadiamond-2026-07-21","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":40.9,"normalizedScore":22.1002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-ifeval-2026-07-21","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":75.8,"normalizedScore":43.2624,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aaagenticindex-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.58,"normalizedScore":2.4381,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-tau2bench-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":22.8,"normalizedScore":23.0071,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-gdpvalaanormalized-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-gdpvalaa-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":217,"normalizedScore":19.0803,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-livecodebench-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":37.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-sweverified-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":42,"normalizedScore":25.5911,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aacodingindex-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.04,"normalizedScore":21.4822,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aascicode-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.4,"normalizedScore":58.1788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-lcr-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29,"normalizedScore":38.3091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-critpt-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-designarenawebsite-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1150,"normalizedScore":61.0561,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-gpqa-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":59.1,"normalizedScore":48.0668,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-mmlupro-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":75.9,"normalizedScore":80.5065,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aagpqadiamond-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":55.7,"normalizedScore":47.9675,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aahle-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.6,"normalizedScore":0.996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aaomniscienceindex-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-41.3,"normalizedScore":36.0283,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-omniscienceaccuracy-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.4,"normalizedScore":38.1443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-omnisciencehallucinationrate-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.4,"normalizedScore":9.1677,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-ifeval-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":86.1,"normalizedScore":73.6998,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aaifbench-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":34.8,"normalizedScore":27.2315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-frontiermathv2tiers13-2026-07-21","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":1.724,"normalizedScore":1.9371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-sweverified-2026-07-21","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":49,"normalizedScore":35.3268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-gpqa-2026-07-21","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":59.4,"normalizedScore":48.4948,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-frontiermathv2tiers13-2026-07-21","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.069,"normalizedScore":2.3247,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-frontiermathv2tier4-2026-07-21","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aaagenticindex-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.1,"normalizedScore":12.7117,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-tau2bench-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":54.1,"normalizedScore":54.5913,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gertlabs-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":42.01,"normalizedScore":34.5731,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gdpvalaanormalized-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.3,"normalizedScore":13.3013,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gdpvalaa-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":665,"normalizedScore":42.759,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-sweverified-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":63.8,"normalizedScore":55.911,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-vibecodebench-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":0.4,"normalizedScore":0.5634,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aacodingindex-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.25,"normalizedScore":36.2323,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aascicode-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42.8,"normalizedScore":70.6577,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-lcr-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66,"normalizedScore":87.1863,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-critpt-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.6,"normalizedScore":8.0495,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aammmupro-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.9,"normalizedScore":83.737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-designarenawebsite-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1197,"normalizedScore":68.8119,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gpqa-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":83,"normalizedScore":82.1658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-hle-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":18.8,"normalizedScore":19.5423,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aagpqadiamond-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.4,"normalizedScore":86.8564,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aaomniscienceindex-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-14.3,"normalizedScore":57.2214,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-omniscienceaccuracy-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39,"normalizedScore":61.512,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.4,"normalizedScore":11.5802,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aaifbench-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":48.7,"normalizedScore":48.2602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-frontiermathv2tiers13-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.138,"normalizedScore":15.8854,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-frontiermathv2tier4-2026-07-21","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-bigcodebench-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"bigcodebench","benchmarkName":"BigCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"BigCodeBench authors","benchmarkVersion":"2026","score":56.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-humaneval-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-humaneval","benchmarkName":"Evaluating Large Language Models Trained on Code","benchmarkCategory":"coding","benchmarkOrganisation":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","benchmarkVersion":"2021","score":69.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-bbh-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":86.9,"normalizedScore":98.2609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-drop-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-drop","benchmarkName":"Discrete Reasoning Over Paragraphs","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88.6,"normalizedScore":99.5495,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-hellaswag-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hellaswag","benchmarkName":"HellaSwag","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-winogrande-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-winogrande","benchmarkName":"WinoGrande","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":79.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-cluewsc-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cluewsc","benchmarkName":"CLUEWSC","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":82.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-longbenchv2-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":44.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-agieval-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-agieval","benchmarkName":"AGIEval","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":82.6,"normalizedScore":96.9136,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmlu-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":88.7,"normalizedScore":73.5043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmluredux-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":89.4,"normalizedScore":72.8711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmlupro-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":68.3,"normalizedScore":69.6927,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmmlu-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":88.8,"normalizedScore":72,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-ceval-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":92.1,"normalizedScore":63.6364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-cmmlu-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmmlu","benchmarkName":"Chinese Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-multiloko-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-multiloko","benchmarkName":"MultiLoKo","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":42.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-simpleqa-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":30.1,"normalizedScore":20.1149,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-supergpqa-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":46.5,"normalizedScore":32.5077,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-factsparametric-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-factsparametric","benchmarkName":"FACTS Parametric","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":33.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-triviaqa-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-triviaqa","benchmarkName":"TriviaQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":82.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mgsm-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mgsm","benchmarkName":"Multilingual Grade School Math","benchmarkCategory":"knowledge","benchmarkOrganisation":"Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","benchmarkVersion":"2022","score":85.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-gsm8k-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gsm8k","benchmarkName":"Grade School Math 8K","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.8,"normalizedScore":72.3077,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mathbenchmark-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mathbenchmark","benchmarkName":"MATH","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":57.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-cmath-2026-07-21","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmath","benchmarkName":"CMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":93.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-tau2bench-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":25.1,"normalizedScore":25.328,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aascicode-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.3,"normalizedScore":54.6374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-lcr-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-critpt-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-designarenawebsite-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":861,"normalizedScore":13.3663,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aagpqadiamond-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":54.3,"normalizedScore":46.0705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aahle-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.3,"normalizedScore":0.3984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aaomniscienceindex-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.7,"normalizedScore":60.0471,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.7,"normalizedScore":28.3505,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.9,"normalizedScore":71.2907,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aaifbench-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":34.3,"normalizedScore":26.475,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-frontiermathv2tiers13-2026-07-21","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0.345,"normalizedScore":0.3876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aaagenticindex-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.1,"normalizedScore":1.5448,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-tau2bench-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":15.5,"normalizedScore":15.6408,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-gdpvalaanormalized-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-gdpvalaa-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":90,"normalizedScore":12.3679,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aacodingindex-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.17,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aascicode-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":17,"normalizedScore":27.1501,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-lcr-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.8,"normalizedScore":34.0819,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-critpt-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aammmupro-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":52.9,"normalizedScore":45.6747,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-designarenawebsite-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":780,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aagpqadiamond-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":58.7,"normalizedScore":52.0325,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aahle-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.3,"normalizedScore":2.3904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aaomniscienceindex-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-52.4,"normalizedScore":27.3155,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-omniscienceaccuracy-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.6,"normalizedScore":19.5876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-omnisciencehallucinationrate-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":78.3,"normalizedScore":22.5573,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aaifbench-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.5,"normalizedScore":34.3419,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-frontiermathv2tiers13-2026-07-21","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aaagenticindex-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.31,"normalizedScore":1.9356,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-tau2bench-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":17.8,"normalizedScore":17.9617,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-gdpvalaanormalized-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-gdpvalaa-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":-16,"normalizedScore":6.7653,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aacodingindex-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.28,"normalizedScore":11.7163,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aascicode-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.1,"normalizedScore":54.3002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-lcr-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46,"normalizedScore":60.7662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-critpt-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aammmupro-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":62.1,"normalizedScore":61.5917,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-designarenawebsite-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":901,"normalizedScore":19.967,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aagpqadiamond-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":67.1,"normalizedScore":63.4146,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aahle-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.8,"normalizedScore":3.3865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aaomniscienceindex-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-41.8,"normalizedScore":35.6358,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-omniscienceaccuracy-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.3,"normalizedScore":36.2543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-omnisciencehallucinationrate-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.3,"normalizedScore":11.7008,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aaifbench-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43,"normalizedScore":39.6369,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-frontiermathv2tiers13-2026-07-21","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0.69,"normalizedScore":0.7753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-tau2bench-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86,"normalizedScore":86.781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-gdpvalaanormalized-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.2,"normalizedScore":3.5256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-gdpvalaa-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":545,"normalizedScore":36.4165,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aaagenticindex-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.25,"normalizedScore":3.6851,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-scicode-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":27,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aacodingindex-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.26,"normalizedScore":24.6894,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aascicode-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":27.1,"normalizedScore":44.1821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-lcr-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25,"normalizedScore":33.0251,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-critpt-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-gpqa-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":59,"normalizedScore":47.9241,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aagpqadiamond-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":59.3,"normalizedScore":52.8455,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aahle-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.2,"normalizedScore":6.1753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aaomniscienceindex-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-65.7,"normalizedScore":16.876,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-omniscienceaccuracy-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.4,"normalizedScore":20.9622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-omnisciencehallucinationrate-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":95.8,"normalizedScore":1.4475,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-ifbench-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":57,"normalizedScore":39.9142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aaifbench-2026-07-21","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":57.4,"normalizedScore":61.4221,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-terminalbench2-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":49.1,"normalizedScore":23.8434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-mcpatlas-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":64,"normalizedScore":58.8737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-toolathlon-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":40.7,"normalizedScore":28.3368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-claweval-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.8,"normalizedScore":73.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-gertlabs-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":54.35,"normalizedScore":60.6509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-sweverified-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.7,"normalizedScore":69.6801,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-swepro-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":49.1,"normalizedScore":16.5775,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-swemultilingual-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":69.7,"normalizedScore":44.0299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-mrcr1m-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":37.5,"normalizedScore":19.1564,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-corpusqa1m-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":15.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-designarenawebsite-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1238,"normalizedScore":75.5776,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-mmlupro-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":83,"normalizedScore":90.609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-simpleqa-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":23.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-chinesesimpleqa-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":71.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-gpqa-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":71.2,"normalizedScore":65.3303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-gpqadiamond-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":71.2,"normalizedScore":65.3303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-hle-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":8.1,"normalizedScore":0.7042,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-hmmtfeb2026-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":40.8,"normalizedScore":21.0821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-imoanswerbench-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":41.9,"normalizedScore":12.0658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-apex-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":1,"normalizedScore":1.3605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-apexshortlist-2026-07-21","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":9.3,"normalizedScore":0.1235,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-terminalbench2-2026-07-21","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":43.1,"normalizedScore":13.1673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-claweval-2026-07-21","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":63.1,"normalizedScore":80.4469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-sweverified-2026-07-21","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":69.4,"normalizedScore":63.6996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-swepro-2026-07-21","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":42.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-swemultilingual-2026-07-21","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":52,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-nl2repo-2026-07-21","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":27.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-bfclv4-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":25.15,"normalizedScore":7.6339,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-livecodebenchpro-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":22.68,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-livecodebenchv6-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":33.52,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-bbh-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":71.89,"normalizedScore":54.7536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-mmlupro-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":48.85,"normalizedScore":42.0176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-mmluredux-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":70.06,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-gpqadiamond-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":26.26,"normalizedScore":1.2127,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-supergpqa-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":23.14,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-ifbench-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":46.67,"normalizedScore":17.7468,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-ifeval-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":80.41,"normalizedScore":56.8853,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-aime2025-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":40.42,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-aime2026-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":40.42,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-hmmtfeb2026-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":25.76,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-math500-2026-07-21","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-math500","benchmarkName":"MATH-500 Problem Set","benchmarkCategory":"mathematics","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2021","score":91.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-terminalbench2-2026-07-21","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":35.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-sweverified-2026-07-21","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":69.9,"normalizedScore":64.395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-swemultilingual-2026-07-21","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":57.7,"normalizedScore":14.1791,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-swepro-2026-07-21","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":46.3,"normalizedScore":9.0909,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-bfclv4-2026-07-21","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":21.08,"normalizedScore":0.0926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-mmmu-2026-07-21","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":32.67,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-realworldqa-2026-07-21","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":58.43,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-countbench-2026-07-21","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-countbench","benchmarkName":"CountBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":73.31,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-gpqa-2026-07-21","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":25.66,"normalizedScore":0.3567,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-mmlupro-2026-07-21","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":19.32,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-ifeval-2026-07-21","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":61.16,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-bfclv4-2026-07-21","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":21.03,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-gpqa-2026-07-21","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":25.41,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-gpqadiamond-2026-07-21","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":25.41,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-mmlupro-2026-07-21","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":20.25,"normalizedScore":1.3233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-ifeval-2026-07-21","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":71.71,"normalizedScore":31.1761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-ifbench-2026-07-21","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":38.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-tau2bench-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":80.7,"normalizedScore":81.4329,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaagenticindex-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.16,"normalizedScore":16.5457,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-gdpvalaanormalized-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.7,"normalizedScore":17.1474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-gdpvalaa-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":714,"normalizedScore":45.3488,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-terminalbenchhard-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":25,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aacodingindex-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.85,"normalizedScore":28.4311,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aascicode-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.8,"normalizedScore":62.226,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-lcr-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46,"normalizedScore":60.7662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-critpt-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-mmmu-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":75.1,"normalizedScore":79.5612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-mmmupro-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":63,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-charxiv-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":52.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aammmupro-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":63.2,"normalizedScore":63.4948,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aagpqadiamond-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.1,"normalizedScore":75.6098,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aahle-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":11.4,"normalizedScore":16.5339,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaomniscienceindex-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-4,"normalizedScore":65.3061,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-omniscienceaccuracy-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.9,"normalizedScore":9.7938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-omnisciencehallucinationrate-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.1,"normalizedScore":100,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaopennessindex-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaifbench-2026-07-21","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.9,"normalizedScore":86.3843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-researchclawbench-2026-07-21","modelSlug":"grok-4-1","modelName":"Grok 4.1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":13.5,"normalizedScore":12.6437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-reasoning-vibecodebench-2026-07-21","modelSlug":"glm-5-reasoning","modelName":"GLM-5 (Reasoning)","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":23.359,"normalizedScore":32.8986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-reasoning-designarenawebsite-2026-07-21","modelSlug":"glm-5-reasoning","modelName":"GLM-5 (Reasoning)","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1278,"normalizedScore":82.1782,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-claweval-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":55.8,"normalizedScore":70.2514,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-tau2bench-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aascicode-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.6,"normalizedScore":72.0067,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-lcr-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.7,"normalizedScore":80.1849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-critpt-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-designarenawebsite-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1301,"normalizedScore":85.9736,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aagpqadiamond-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.7,"normalizedScore":87.2629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aahle-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":25.4,"normalizedScore":44.4223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aaomniscienceindex-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-15.1,"normalizedScore":56.5934,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-omniscienceaccuracy-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29,"normalizedScore":44.3299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-omnisciencehallucinationrate-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.2,"normalizedScore":41.9783,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aaifbench-2026-07-21","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.2,"normalizedScore":85.3253,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-claweval-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.8,"normalizedScore":73.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-tau2bench-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95,"normalizedScore":95.8628,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-gertlabs-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":36.68,"normalizedScore":23.3094,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-researchclawbench-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":15.3,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-sweverified-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":78,"normalizedScore":75.6606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aascicode-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42.5,"normalizedScore":70.1518,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-lcr-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.7,"normalizedScore":80.1849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-critpt-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aagpqadiamond-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87,"normalizedScore":90.3794,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aahle-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.3,"normalizedScore":50.1992,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aaomniscienceindex-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.9,"normalizedScore":72.292,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-omniscienceaccuracy-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":40.5498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-omnisciencehallucinationrate-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.9,"normalizedScore":80.9409,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aaifbench-2026-07-21","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":68.8,"normalizedScore":78.6687,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-claweval-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":53.8,"normalizedScore":67.4581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-tau2bench-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-gertlabs-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":30.76,"normalizedScore":10.7988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aascicode-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.5,"normalizedScore":71.8381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-lcr-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61,"normalizedScore":80.5812,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-critpt-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aammmupro-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.8,"normalizedScore":80.1038,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-designarenawebsite-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1258,"normalizedScore":78.8779,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aagpqadiamond-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":80.9,"normalizedScore":82.1138,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aahle-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":15.8,"normalizedScore":25.2988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aaomniscienceindex-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-19,"normalizedScore":53.5322,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-omniscienceaccuracy-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.1,"normalizedScore":44.5017,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-omnisciencehallucinationrate-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.9,"normalizedScore":35.1025,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aaifbench-2026-07-21","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":61.1,"normalizedScore":67.0197,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aaagenticindex-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.71,"normalizedScore":47.3479,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-tau2bench-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.7,"normalizedScore":45.9936,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-gdpvalaa-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1075,"normalizedScore":64.4292,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-jobbench-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":8.53,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-vibecodebench-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":20.088,"normalizedScore":28.2918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aacodingindex-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.78,"normalizedScore":42.7767,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aascicode-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42.9,"normalizedScore":70.8263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-lcr-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.6,"normalizedScore":99.8679,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-critpt-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.7,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aammmupro-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.2,"normalizedScore":82.526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-designarenawebsite-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1214,"normalizedScore":71.6172,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aagpqadiamond-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.4,"normalizedScore":88.2114,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aahle-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":26.5,"normalizedScore":46.6135,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-8.1,"normalizedScore":62.0879,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.7,"normalizedScore":64.433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82.1,"normalizedScore":17.9735,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aaifbench-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.1,"normalizedScore":85.174,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aamath500-2026-07-21","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aamath500","benchmarkName":"Artificial Analysis MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-tau2bench-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.3,"normalizedScore":94.1473,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-vibecodebench-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":1.2,"normalizedScore":1.6901,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aascicode-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":44.2,"normalizedScore":73.0185,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-lcr-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68,"normalizedScore":89.8283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-critpt-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.9,"normalizedScore":8.9783,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aammmupro-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":63.3,"normalizedScore":63.6678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aagpqadiamond-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.3,"normalizedScore":88.0759,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aahle-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":17.6,"normalizedScore":28.8845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aaomniscienceindex-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-28.7,"normalizedScore":45.9184,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-omniscienceaccuracy-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.3,"normalizedScore":37.9725,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-omnisciencehallucinationrate-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.4,"normalizedScore":29.6743,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aaifbench-2026-07-21","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":52.7,"normalizedScore":54.3116,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-claweval-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":45.2,"normalizedScore":55.4469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-tau2bench-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":91.2,"normalizedScore":92.0283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-sweverified-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.8,"normalizedScore":71.21,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aascicode-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.7,"normalizedScore":60.371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-lcr-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-critpt-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aammmupro-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":69.9,"normalizedScore":75.0865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aagpqadiamond-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":82.8,"normalizedScore":84.6883,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aahle-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":19.9,"normalizedScore":33.4661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aaomniscienceindex-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-17.4,"normalizedScore":54.7881,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-omniscienceaccuracy-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.7,"normalizedScore":26.6323,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-omnisciencehallucinationrate-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.4,"normalizedScore":63.4499,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aaifbench-2026-07-21","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":53.5,"normalizedScore":55.5219,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-thinking-vibecodebench-2026-07-21","modelSlug":"deepseek-v3-2-thinking","modelName":"DeepSeek V3.2 (Thinking)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":5.108,"normalizedScore":7.1941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-thinking-designarenawebsite-2026-07-21","modelSlug":"deepseek-v3-2-thinking","modelName":"DeepSeek V3.2 (Thinking)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1204,"normalizedScore":69.967,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-tau2bench-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":63.7,"normalizedScore":64.2785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-gertlabs-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":47.32,"normalizedScore":45.7946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aascicode-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.6,"normalizedScore":48.398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-lcr-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22,"normalizedScore":29.0621,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-critpt-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aammmupro-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":48.4,"normalizedScore":37.8893,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aagpqadiamond-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":63.7,"normalizedScore":58.8076,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aahle-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5,"normalizedScore":3.7849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aaomniscienceindex-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-50.9,"normalizedScore":28.4929,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-omniscienceaccuracy-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17,"normalizedScore":23.7113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-omnisciencehallucinationrate-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.8,"normalizedScore":18.3353,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aaifbench-2026-07-21","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.5,"normalizedScore":29.8033,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-tau2bench-2026-07-21","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":34.8,"normalizedScore":35.116,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aascicode-2026-07-21","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.7,"normalizedScore":60.371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-lcr-2026-07-21","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45,"normalizedScore":59.4452,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-critpt-2026-07-21","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-designarenawebsite-2026-07-21","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1152,"normalizedScore":61.3861,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aagpqadiamond-2026-07-21","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":73.5,"normalizedScore":72.0867,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aahle-2026-07-21","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.3,"normalizedScore":6.3745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aaomniscienceindex-2026-07-21","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-41.1,"normalizedScore":36.1852,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-omniscienceaccuracy-2026-07-21","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.1,"normalizedScore":34.1924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-omnisciencehallucinationrate-2026-07-21","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.5,"normalizedScore":16.2847,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aaifbench-2026-07-21","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":37.8,"normalizedScore":31.77,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aaagenticindex-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.52,"normalizedScore":9.7711,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-tau2bench-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":24.6,"normalizedScore":24.8234,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-gdpvalaanormalized-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.6,"normalizedScore":10.5769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-gdpvalaa-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":633,"normalizedScore":41.0677,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aacodingindex-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.07,"normalizedScore":17.1916,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aascicode-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.2,"normalizedScore":59.5278,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-lcr-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.7,"normalizedScore":45.8388,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-critpt-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aammmupro-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":55.7,"normalizedScore":50.519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aagpqadiamond-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68,"normalizedScore":64.6341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aahle-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.1,"normalizedScore":1.992,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aaomniscienceindex-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-39.4,"normalizedScore":37.5196,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-omniscienceaccuracy-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.1,"normalizedScore":35.9107,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-omnisciencehallucinationrate-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.7,"normalizedScore":16.0434,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aaifbench-2026-07-21","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.2,"normalizedScore":29.3495,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-designarenawebsite-2026-07-21","modelSlug":"glm-4-5","modelName":"GLM-4.5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1200,"normalizedScore":69.3069,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-tau2bench-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":65.8,"normalizedScore":66.3976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-vibecodebench-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aascicode-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":44.2,"normalizedScore":73.0185,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-lcr-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.7,"normalizedScore":85.469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-critpt-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.9,"normalizedScore":8.9783,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aammmupro-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":61.8,"normalizedScore":61.0727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aagpqadiamond-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.7,"normalizedScore":87.2629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aahle-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":17,"normalizedScore":27.6892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aaomniscienceindex-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-28.4,"normalizedScore":46.1538,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-omniscienceaccuracy-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.6,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-omnisciencehallucinationrate-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66,"normalizedScore":37.3945,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aaifbench-2026-07-21","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":50.5,"normalizedScore":50.9834,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-tau2bench-2026-07-21","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":36.5,"normalizedScore":36.8315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aascicode-2026-07-21","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.3,"normalizedScore":66.4418,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-lcr-2026-07-21","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.7,"normalizedScore":72.2589,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-critpt-2026-07-21","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aagpqadiamond-2026-07-21","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81.3,"normalizedScore":82.6558,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aahle-2026-07-21","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.9,"normalizedScore":23.506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aaomniscienceindex-2026-07-21","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-27.1,"normalizedScore":47.1743,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-omniscienceaccuracy-2026-07-21","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31,"normalizedScore":47.7663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-omnisciencehallucinationrate-2026-07-21","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84,"normalizedScore":15.6815,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aaifbench-2026-07-21","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.6,"normalizedScore":34.4932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-5-vibecodebench-2026-07-21","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":14.852,"normalizedScore":20.9174,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-preview-aacodingindex-2026-07-21","modelSlug":"o1-preview","modelName":"o1-preview","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.05,"normalizedScore":37.388,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1-preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-tau2bench-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":90.1,"normalizedScore":90.9183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-gdpvalaanormalized-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.7,"normalizedScore":4.3269,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-gdpvalaa-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":554,"normalizedScore":36.8922,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aascicode-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.1,"normalizedScore":59.3592,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-lcr-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33,"normalizedScore":43.5931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-critpt-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-designarenawebsite-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1165,"normalizedScore":63.5314,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-mmlu-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":87.2,"normalizedScore":60.6838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-mmluproarcee-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-gpqadiamond-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":63.3,"normalizedScore":54.0591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aahle-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.7,"normalizedScore":23.1076,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aaomniscienceindex-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-44.2,"normalizedScore":33.752,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-omniscienceaccuracy-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.8,"normalizedScore":33.677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-omnisciencehallucinationrate-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.6,"normalizedScore":12.5452,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aaifbench-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":56.3,"normalizedScore":59.7579,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aime2025arcee-2026-07-21","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":24,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-tau2bench-2026-07-21","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":46.5,"normalizedScore":46.9223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aascicode-2026-07-21","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":30.6,"normalizedScore":50.0843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-lcr-2026-07-21","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.7,"normalizedScore":57.7279,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-critpt-2026-07-21","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-designarenawebsite-2026-07-21","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1176,"normalizedScore":65.3465,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aagpqadiamond-2026-07-21","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":73.3,"normalizedScore":71.8157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aahle-2026-07-21","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.8,"normalizedScore":7.3705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aaomniscienceindex-2026-07-21","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-62.5,"normalizedScore":19.3878,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-omniscienceaccuracy-2026-07-21","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.5,"normalizedScore":21.134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-omnisciencehallucinationrate-2026-07-21","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":92.3,"normalizedScore":5.6695,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aaifbench-2026-07-21","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":37.6,"normalizedScore":31.4675,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-tau2bench-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":90.1,"normalizedScore":90.9183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gertlabs-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":32.55,"normalizedScore":14.5816,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gdpvalaanormalized-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.7,"normalizedScore":4.3269,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gdpvalaa-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":554,"normalizedScore":36.8922,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-sweverifiedarcee-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":63.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aascicode-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.1,"normalizedScore":59.3592,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-lcr-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33,"normalizedScore":43.5931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-critpt-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-designarenawebsite-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1165,"normalizedScore":63.5314,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gpqadiamond-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":76.3,"normalizedScore":72.6066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-mmluproarcee-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":83.4,"normalizedScore":58.9928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aahle-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.7,"normalizedScore":23.1076,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aaomniscienceindex-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-44.2,"normalizedScore":33.752,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-omniscienceaccuracy-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.8,"normalizedScore":33.677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-omnisciencehallucinationrate-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.6,"normalizedScore":12.5452,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aaifbench-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":56.3,"normalizedScore":59.7579,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aime2025arcee-2026-07-21","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":96.3,"normalizedScore":95.3826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aaagenticindex-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.27,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-tau2bench-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":10.5,"normalizedScore":10.5954,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-gdpvalaanormalized-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-gdpvalaa-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":-144,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aacodingindex-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.06,"normalizedScore":2.7304,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aascicode-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":21.2,"normalizedScore":34.2327,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-lcr-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.7,"normalizedScore":7.5297,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-critpt-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aammmupro-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":48,"normalizedScore":37.1972,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aagpqadiamond-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":42.8,"normalizedScore":30.4878,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aahle-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.7,"normalizedScore":3.1873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aaomniscienceindex-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-65.9,"normalizedScore":16.719,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-omniscienceaccuracy-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":12.5,"normalizedScore":15.9794,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.5,"normalizedScore":9.047,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aaifbench-2026-07-21","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":31.8,"normalizedScore":22.6929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aaagenticindex-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.7,"normalizedScore":8.2449,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-tau2bench-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":41.2,"normalizedScore":41.5742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-gdpvalaanormalized-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.4,"normalizedScore":7.0513,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-gdpvalaa-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":588,"normalizedScore":38.6892,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aacodingindex-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.64,"normalizedScore":26.683,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aascicode-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38,"normalizedScore":62.5632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-lcr-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.7,"normalizedScore":59.0489,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-critpt-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aammmupro-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":56.8,"normalizedScore":52.4221,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aagpqadiamond-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.9,"normalizedScore":76.6938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aahle-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":9.5,"normalizedScore":12.749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aaomniscienceindex-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-29.9,"normalizedScore":44.9765,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-omniscienceaccuracy-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.1,"normalizedScore":32.4742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-omnisciencehallucinationrate-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.8,"normalizedScore":36.4294,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aaifbench-2026-07-21","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":48.2,"normalizedScore":47.5038,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaagenticindex-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.17,"normalizedScore":24.0089,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-apexagentsaa-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":3.1,"normalizedScore":5.1724,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-tau2bench-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":65.8,"normalizedScore":66.3976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15,"normalizedScore":24.0385,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-gdpvalaa-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":799,"normalizedScore":49.8414,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-gertlabs-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":29.61,"normalizedScore":8.3686,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaenterpriseopsgym-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaitbench-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-reactnativeevals-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71.6,"normalizedScore":2.3904,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aacodingindex-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.44,"normalizedScore":32.1728,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aascicode-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.9,"normalizedScore":64.0809,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aalivecodebench-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aalivecodebench","benchmarkName":"Artificial Analysis LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-lcr-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.7,"normalizedScore":66.9749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-critpt-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-designarenawebsite-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":998,"normalizedScore":35.9736,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aagpqadiamond-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":78.2,"normalizedScore":78.4553,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aahle-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":18.5,"normalizedScore":30.6773,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaomniscienceindex-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-50,"normalizedScore":29.1994,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.5,"normalizedScore":31.4433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.2,"normalizedScore":6.9964,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaopennessindex-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aammlupro-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaglobalmmlulite-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaifbench-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":69,"normalizedScore":78.9713,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaaime2025-2026-07-21","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaaime2025","benchmarkName":"Artificial Analysis AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-tau2bench-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83,"normalizedScore":83.7538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-vibecodebench-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":22.168,"normalizedScore":31.2212,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aascicode-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.2,"normalizedScore":66.2732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-lcr-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.3,"normalizedScore":88.9036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-critpt-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.7,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aammmupro-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.5,"normalizedScore":79.5848,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aagpqadiamond-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86,"normalizedScore":89.0244,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aahle-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.4,"normalizedScore":40.4382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-6,"normalizedScore":63.7363,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.2,"normalizedScore":61.8557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.4,"normalizedScore":27.2618,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aaifbench-2026-07-21","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70,"normalizedScore":80.4841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-tau2bench-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":92.1,"normalizedScore":92.9364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-gertlabs-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":51.79,"normalizedScore":55.2409,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-jobbench-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":26,"normalizedScore":37.8384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-vibecodebench-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":37.912,"normalizedScore":53.3949,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aascicode-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":54.6,"normalizedScore":90.5565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-lcr-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-critpt-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":8.7,"normalizedScore":26.935,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aammmupro-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":76.3,"normalizedScore":86.1592,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aagpqadiamond-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.9,"normalizedScore":94.3089,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aahle-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":33.5,"normalizedScore":60.5578,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-2.5,"normalizedScore":66.4835,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.7,"normalizedScore":64.433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.8,"normalizedScore":29.1918,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aaifbench-2026-07-21","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":77.6,"normalizedScore":91.9818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-tau2bench-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86.5,"normalizedScore":87.2856,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aascicode-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":41.1,"normalizedScore":67.7909,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-lcr-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.8,"normalizedScore":96.1691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-critpt-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aammmupro-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.3,"normalizedScore":82.699,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-designarenawebsite-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1214,"normalizedScore":71.6172,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aagpqadiamond-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.2,"normalizedScore":86.5854,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aahle-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.5,"normalizedScore":40.6375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.1,"normalizedScore":60.5181,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":61.3402,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.1,"normalizedScore":20.386,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aaifbench-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.6,"normalizedScore":81.3918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aamath500-2026-07-21","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aamath500","benchmarkName":"Artificial Analysis MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aacodingindex-2026-07-21","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.53,"normalizedScore":16.4114,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aascicode-2026-07-21","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":23.3,"normalizedScore":37.774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aagpqadiamond-2026-07-21","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":48.9,"normalizedScore":38.7534,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aahle-2026-07-21","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-tau2bench-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":40.9,"normalizedScore":41.2714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-gdpvalaanormalized-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-gdpvalaa-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":484,"normalizedScore":33.1924,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aaagenticindex-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.99,"normalizedScore":3.2012,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aascicode-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.6,"normalizedScore":48.398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aacodingindex-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.37,"normalizedScore":8.9569,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-lcr-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.7,"normalizedScore":44.5178,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-critpt-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aagpqadiamond-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":75.7,"normalizedScore":75.0678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aahle-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":10.2,"normalizedScore":14.1434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aaomniscienceindex-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-51.6,"normalizedScore":27.9435,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-omniscienceaccuracy-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.1,"normalizedScore":23.8832,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-omnisciencehallucinationrate-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82.9,"normalizedScore":17.0084,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aaifbench-2026-07-21","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":71.1,"normalizedScore":82.1483,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aaagenticindex-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.1,"normalizedScore":5.2671,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-apexagentsaa-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":0.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-tau2bench-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":60.2,"normalizedScore":60.7467,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3,"normalizedScore":4.8077,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-gdpvalaa-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":559,"normalizedScore":37.1564,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-reactnativeevals-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aacodingindex-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.7,"normalizedScore":18.1017,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aascicode-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":34.4,"normalizedScore":56.4924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-lcr-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.7,"normalizedScore":40.5548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-critpt-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-designarenawebsite-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":882,"normalizedScore":16.8317,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aagpqadiamond-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68.8,"normalizedScore":65.7182,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aahle-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":9.8,"normalizedScore":13.3466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aaomniscienceindex-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-63.9,"normalizedScore":18.2889,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.5,"normalizedScore":21.134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94.1,"normalizedScore":3.4982,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aaifbench-2026-07-21","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":65.1,"normalizedScore":73.0711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aaagenticindex-2026-07-21","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.96,"normalizedScore":1.2842,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-gdpvalaanormalized-2026-07-21","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-gdpvalaa-2026-07-21","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":226,"normalizedScore":19.556,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aascicode-2026-07-21","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":22.9,"normalizedScore":37.0995,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aacodingindex-2026-07-21","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":11.38,"normalizedScore":4.6374,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aammmupro-2026-07-21","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":41.5,"normalizedScore":25.9516,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aagpqadiamond-2026-07-21","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":42.6,"normalizedScore":30.2168,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aahle-2026-07-21","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4,"normalizedScore":1.7928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aaifbench-2026-07-21","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":31,"normalizedScore":21.4826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-tau2bench-2026-07-21","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":30.7,"normalizedScore":30.9788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aascicode-2026-07-21","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.2,"normalizedScore":47.7234,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-lcr-2026-07-21","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.3,"normalizedScore":7.0013,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-critpt-2026-07-21","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aagpqadiamond-2026-07-21","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":48.6,"normalizedScore":38.3469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aahle-2026-07-21","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4,"normalizedScore":1.7928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aaomniscienceindex-2026-07-21","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-34,"normalizedScore":41.7582,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-omniscienceaccuracy-2026-07-21","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.1,"normalizedScore":29.0378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-omnisciencehallucinationrate-2026-07-21","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.8,"normalizedScore":35.2232,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aaifbench-2026-07-21","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":31.2,"normalizedScore":21.7852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-tau2bench-2026-07-21","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":19,"normalizedScore":19.1726,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aascicode-2026-07-21","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.9,"normalizedScore":48.9039,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-lcr-2026-07-21","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.3,"normalizedScore":32.1004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-critpt-2026-07-21","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aagpqadiamond-2026-07-21","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":51.5,"normalizedScore":42.2764,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aahle-2026-07-21","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.2,"normalizedScore":2.1912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aaomniscienceindex-2026-07-21","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-17.3,"normalizedScore":54.8666,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-omniscienceaccuracy-2026-07-21","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.3,"normalizedScore":32.8179,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-omnisciencehallucinationrate-2026-07-21","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51,"normalizedScore":55.4885,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aaifbench-2026-07-21","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39,"normalizedScore":33.5855,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen2-5-coder-32b-instruct-aascicode-2026-07-21","modelSlug":"qwen2-5-coder-32b-instruct","modelName":"Qwen2.5 Coder 32B Instruct","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":27.1,"normalizedScore":44.1821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen2-5-coder-32b-instruct-aagpqadiamond-2026-07-21","modelSlug":"qwen2-5-coder-32b-instruct","modelName":"Qwen2.5 Coder 32B Instruct","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":41.7,"normalizedScore":28.9973,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen2-5-coder-32b-instruct-aahle-2026-07-21","modelSlug":"qwen2-5-coder-32b-instruct","modelName":"Qwen2.5 Coder 32B Instruct","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.8,"normalizedScore":1.3944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-super-100b-claweval-2026-07-21","modelSlug":"nemotron-3-super-100b","modelName":"Nemotron 3 Super 100B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":5.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Super 100B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aacodingindex-2026-07-21","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.63,"normalizedScore":22.3346,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aascicode-2026-07-21","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.5,"normalizedScore":48.2293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aammmupro-2026-07-21","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":55,"normalizedScore":49.308,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aagpqadiamond-2026-07-21","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":58.9,"normalizedScore":52.3035,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aahle-2026-07-21","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.9,"normalizedScore":3.5857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-tau2bench-2026-07-21","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aascicode-2026-07-21","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":26,"normalizedScore":42.3272,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-lcr-2026-07-21","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-critpt-2026-07-21","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aagpqadiamond-2026-07-21","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":57.5,"normalizedScore":50.4065,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aahle-2026-07-21","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.1,"normalizedScore":1.992,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aaomniscienceindex-2026-07-21","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-56.7,"normalizedScore":23.9403,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-omniscienceaccuracy-2026-07-21","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.2,"normalizedScore":17.1821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-omnisciencehallucinationrate-2026-07-21","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.5,"normalizedScore":19.9035,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aaifbench-2026-07-21","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":23.5,"normalizedScore":10.1362,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-mini-vibecodebench-2026-07-21","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":14.171,"normalizedScore":19.9583,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-pro-aagpqadiamond-2026-07-21","modelSlug":"o3-pro","modelName":"o3-pro","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.5,"normalizedScore":86.9919,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-tau2bench-2026-07-21","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":75.7,"normalizedScore":76.3875,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-sweverified-2026-07-21","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":70.8,"normalizedScore":65.6467,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aascicode-2026-07-21","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.2,"normalizedScore":59.5278,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-lcr-2026-07-21","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.3,"normalizedScore":63.8045,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-critpt-2026-07-21","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aagpqadiamond-2026-07-21","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":72.7,"normalizedScore":71.0027,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aahle-2026-07-21","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7.5,"normalizedScore":8.7649,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aaomniscienceindex-2026-07-21","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-36,"normalizedScore":40.1884,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-omniscienceaccuracy-2026-07-21","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.8,"normalizedScore":35.3952,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-omnisciencehallucinationrate-2026-07-21","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":78.5,"normalizedScore":22.316,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aaifbench-2026-07-21","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":41.4,"normalizedScore":37.2163,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-turbo-aacodingindex-2026-07-21","modelSlug":"gpt-4-turbo","modelName":"GPT-4 Turbo","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.49,"normalizedScore":19.243,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-turbo-aascicode-2026-07-21","modelSlug":"gpt-4-turbo","modelName":"GPT-4 Turbo","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":31.9,"normalizedScore":52.2766,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-turbo-aahle-2026-07-21","modelSlug":"gpt-4-turbo","modelName":"GPT-4 Turbo","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.3,"normalizedScore":0.3984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-tau2bench-2026-07-21","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":11.4,"normalizedScore":11.5035,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aascicode-2026-07-21","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":34.7,"normalizedScore":56.9983,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-lcr-2026-07-21","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.3,"normalizedScore":9.6433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-critpt-2026-07-21","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aagpqadiamond-2026-07-21","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":72.8,"normalizedScore":71.1382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aahle-2026-07-21","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":8.1,"normalizedScore":9.9602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aaomniscienceindex-2026-07-21","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-45.5,"normalizedScore":32.7316,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-omniscienceaccuracy-2026-07-21","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.9,"normalizedScore":28.6942,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-omnisciencehallucinationrate-2026-07-21","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.7,"normalizedScore":18.456,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aaifbench-2026-07-21","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":38.2,"normalizedScore":32.3752,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-0-pro-aascicode-2026-07-21","modelSlug":"gemini-1-0-pro","modelName":"Gemini 1.0 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":11.7,"normalizedScore":18.2125,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-0-pro-aagpqadiamond-2026-07-21","modelSlug":"gemini-1-0-pro","modelName":"Gemini 1.0 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":27.7,"normalizedScore":10.0271,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-0-pro-aahle-2026-07-21","modelSlug":"gemini-1-0-pro","modelName":"Gemini 1.0 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-tau2bench-2026-07-21","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":21.1,"normalizedScore":21.2916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aascicode-2026-07-21","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":18.6,"normalizedScore":29.8482,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-lcr-2026-07-21","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21,"normalizedScore":27.7411,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-critpt-2026-07-21","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aammmupro-2026-07-21","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":30.8,"normalizedScore":7.4394,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aagpqadiamond-2026-07-21","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":37.4,"normalizedScore":23.1707,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aahle-2026-07-21","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.9,"normalizedScore":1.5936,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aaomniscienceindex-2026-07-21","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-47.6,"normalizedScore":31.0832,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-omniscienceaccuracy-2026-07-21","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.2,"normalizedScore":24.055,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-omnisciencehallucinationrate-2026-07-21","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":78.2,"normalizedScore":22.6779,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aaifbench-2026-07-21","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.1,"normalizedScore":29.1982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-tau2bench-2026-07-21","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":71.4,"normalizedScore":72.0484,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aascicode-2026-07-21","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.9,"normalizedScore":67.4536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-lcr-2026-07-21","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.3,"normalizedScore":87.5826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-critpt-2026-07-21","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aammmupro-2026-07-21","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":67.9,"normalizedScore":71.6263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aagpqadiamond-2026-07-21","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":80.9,"normalizedScore":82.1138,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aahle-2026-07-21","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":11.9,"normalizedScore":17.5299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aaifbench-2026-07-21","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":55.4,"normalizedScore":58.3964,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-tau2bench-2026-07-21","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":14,"normalizedScore":14.1271,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aascicode-2026-07-21","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":20.8,"normalizedScore":33.5582,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-lcr-2026-07-21","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19,"normalizedScore":25.0991,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-critpt-2026-07-21","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aammmupro-2026-07-21","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":44.3,"normalizedScore":30.7958,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aagpqadiamond-2026-07-21","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":49.9,"normalizedScore":40.1084,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aahle-2026-07-21","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.4,"normalizedScore":0.5976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aaomniscienceindex-2026-07-21","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-47.6,"normalizedScore":31.0832,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-omniscienceaccuracy-2026-07-21","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17,"normalizedScore":23.7113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-omnisciencehallucinationrate-2026-07-21","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.9,"normalizedScore":23.0398,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aaifbench-2026-07-21","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":38.1,"normalizedScore":32.2239,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-tau2bench-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":8.5,"normalizedScore":8.5772,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aascicode-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":3,"normalizedScore":3.5413,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-lcr-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-critpt-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractjsonvalidity-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractjsonvalidity","benchmarkName":"Liquid image-to-JSON extraction JSON validity","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":99.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractschemaf1-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractschemaf1","benchmarkName":"Liquid image-to-JSON extraction schema consistency F1","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":99.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractvlmjudge-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractvlmjudge","benchmarkName":"Liquid image-to-JSON extraction VLM judge score","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":90.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aammmupro-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":26.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aagpqadiamond-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":28.9,"normalizedScore":11.6531,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aahle-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.1,"normalizedScore":3.9841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aaomniscienceindex-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-83.9,"normalizedScore":2.5903,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-omniscienceaccuracy-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.2,"normalizedScore":3.4364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-omnisciencehallucinationrate-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94,"normalizedScore":3.6188,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aaifbench-2026-07-21","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":33.1,"normalizedScore":24.6596,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-extract-liquidextractjsonvalidity-2026-07-21","modelSlug":"lfm2-5-vl-450m-extract","modelName":"LFM2.5-VL-450M-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractjsonvalidity","benchmarkName":"Liquid image-to-JSON extraction JSON validity","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":98.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-extract-liquidextractschemaf1-2026-07-21","modelSlug":"lfm2-5-vl-450m-extract","modelName":"LFM2.5-VL-450M-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractschemaf1","benchmarkName":"Liquid image-to-JSON extraction schema consistency F1","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":98.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-extract-liquidextractvlmjudge-2026-07-21","modelSlug":"lfm2-5-vl-450m-extract","modelName":"LFM2.5-VL-450M-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractvlmjudge","benchmarkName":"Liquid image-to-JSON extraction VLM judge score","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":84.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-cyber-cybergym-2026-07-21","modelSlug":"sakana-fugu-cyber","modelName":"Fugu Cyber","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":86.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-cyber-ctirealm-2026-07-21","modelSlug":"sakana-fugu-cyber","modelName":"Fugu Cyber","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-ctirealm","benchmarkName":"CTI-REALM","benchmarkCategory":"agents","benchmarkOrganisation":"Sakana AI","benchmarkVersion":"2026","score":72.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-kimiclaw247-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-kimiclaw247","benchmarkName":"Kimi Claw 24/7 Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":46.9,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-mcpatlas-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":76,"normalizedScore":79.3515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-mcpmarkverified-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mcpmarkverified","benchmarkName":"MCPMark-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"MCPMark","benchmarkVersion":"2026","score":81.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aaagenticindex-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.59,"normalizedScore":54.5691,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-tau2bench-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":90.1,"normalizedScore":90.9183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-gdpvalaanormalized-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.3,"normalizedScore":54.9679,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-gdpvalaa-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1187,"normalizedScore":70.3488,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-kimicodebenchv2-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"kimi-code-bench-v2","benchmarkName":"Kimi Code Bench v2","benchmarkCategory":"coding","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":62,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-programbench-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-programbench","benchmarkName":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","benchmarkCategory":"coding","benchmarkOrganisation":"John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","benchmarkVersion":"2026","score":53.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-mlsbenchlite-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mlsbenchlite","benchmarkName":"MLS-Bench Lite","benchmarkCategory":"coding","benchmarkOrganisation":"MLS-Bench","benchmarkVersion":"2026","score":35.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aacodingindex-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.76,"normalizedScore":75.9752,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aascicode-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.5,"normalizedScore":78.5835,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-lcr-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.3,"normalizedScore":87.5826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-critpt-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":10,"normalizedScore":30.9598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-designarenawebsite-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1302,"normalizedScore":86.1386,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aagpqadiamond-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.6,"normalizedScore":93.9024,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aahle-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":32.8,"normalizedScore":59.1633,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aaomniscienceindex-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.7,"normalizedScore":60.0471,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-omniscienceaccuracy-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.6,"normalizedScore":60.8247,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-omnisciencehallucinationrate-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.3,"normalizedScore":20.1448,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aaifbench-2026-07-21","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":63.1,"normalizedScore":70.0454,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-tau2bench-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83,"normalizedScore":83.7538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-gertlabs-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":49.68,"normalizedScore":50.7819,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-jobbench-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":26.2,"normalizedScore":38.2716,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-vibecodebench-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":13.115,"normalizedScore":18.4711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aascicode-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.2,"normalizedScore":66.2732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-lcr-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.3,"normalizedScore":88.9036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-critpt-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.7,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aammmupro-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.5,"normalizedScore":79.5848,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-designarenawebsite-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1191,"normalizedScore":67.8218,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aagpqadiamond-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86,"normalizedScore":89.0244,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aahle-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.4,"normalizedScore":40.4382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aaomniscienceindex-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-6,"normalizedScore":63.7363,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-omniscienceaccuracy-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.2,"normalizedScore":61.8557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-omnisciencehallucinationrate-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.4,"normalizedScore":27.2618,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aaifbench-2026-07-21","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70,"normalizedScore":80.4841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-1-35b-a3b-androidworld-2026-07-21","modelSlug":"holo3-1-35b-a3b","modelName":"Holo3.1-35B-A3B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":79.3,"normalizedScore":84.1121,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3.1-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-1-4b-androidworld-2026-07-21","modelSlug":"holo3-1-4b","modelName":"Holo3.1-4B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":71,"normalizedScore":6.5421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3.1-4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-1-9b-androidworld-2026-07-21","modelSlug":"holo3-1-9b","modelName":"Holo3.1-9B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":71,"normalizedScore":6.5421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3.1-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-tau2bench-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74.3,"normalizedScore":74.9748,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-gertlabs-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":43.74,"normalizedScore":38.2291,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-vibecodebench-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":3.506,"normalizedScore":4.9378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aascicode-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.3,"normalizedScore":63.0691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-lcr-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.7,"normalizedScore":61.6909,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-critpt-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-designarenawebsite-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1148,"normalizedScore":60.7261,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aagpqadiamond-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.4,"normalizedScore":76.0163,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aahle-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":11.1,"normalizedScore":15.9363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aaomniscienceindex-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-43.1,"normalizedScore":34.6154,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-omniscienceaccuracy-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.4,"normalizedScore":36.4261,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-omnisciencehallucinationrate-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.4,"normalizedScore":9.1677,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aaifbench-2026-07-21","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":44.1,"normalizedScore":41.3011,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-fast-reactnativeevals-2026-07-21","modelSlug":"composer-2-fast","modelName":"Composer 2 Fast","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":94.9,"normalizedScore":95.2191,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-tau2bench-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":89.5,"normalizedScore":90.3128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-vibecodebench-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":20.63,"normalizedScore":29.0551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aascicode-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49.5,"normalizedScore":81.9562,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-lcr-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-critpt-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.6,"normalizedScore":14.2415,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aammmupro-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74,"normalizedScore":82.1799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-designarenawebsite-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1277,"normalizedScore":82.0132,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aagpqadiamond-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86.6,"normalizedScore":89.8374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aahle-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.4,"normalizedScore":50.3984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aaomniscienceindex-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.3,"normalizedScore":78.8854,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-omniscienceaccuracy-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.7,"normalizedScore":73.0241,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-omnisciencehallucinationrate-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59.8,"normalizedScore":44.8733,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aammlupro-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.5,"normalizedScore":96.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aaglobalmmlulite-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.3,"normalizedScore":81.7308,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aaifbench-2026-07-21","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":58,"normalizedScore":62.3298,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-thinking-vibecodebench-2026-07-21","modelSlug":"claude-haiku-4-5-thinking","modelName":"Claude Haiku 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":11.393,"normalizedScore":16.0458,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-thinking-designarenawebsite-2026-07-21","modelSlug":"claude-haiku-4-5-thinking","modelName":"Claude Haiku 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1152,"normalizedScore":61.3861,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-thinking-vibecodebench-2026-07-21","modelSlug":"claude-sonnet-4-5-thinking","modelName":"Claude Sonnet 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":22.621,"normalizedScore":31.8592,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-thinking-designarenawebsite-2026-07-21","modelSlug":"claude-sonnet-4-5-thinking","modelName":"Claude Sonnet 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1219,"normalizedScore":72.4422,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-235b-a22b-screenspotpro-2026-07-21","modelSlug":"holo2-235b-a22b","modelName":"Holo2-235B-A22B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":70.6,"normalizedScore":59.0047,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-235B-A22B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-30b-a3b-screenspotpro-2026-07-21","modelSlug":"holo2-30b-a3b","modelName":"Holo2-30B-A3B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":66.1,"normalizedScore":48.3412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-4b-screenspotpro-2026-07-21","modelSlug":"holo2-4b","modelName":"Holo2-4B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":57.2,"normalizedScore":27.2512,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-8b-screenspotpro-2026-07-21","modelSlug":"holo2-8b","modelName":"Holo2-8B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":58.9,"normalizedScore":31.2796,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aaagenticindex-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.73,"normalizedScore":56.6909,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-gdpvalaanormalized-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.7,"normalizedScore":57.2115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-gdpvalaa-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1214,"normalizedScore":71.7759,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aascicode-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.6,"normalizedScore":78.7521,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aacodingindex-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.8,"normalizedScore":73.1436,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-lcr-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-critpt-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.9,"normalizedScore":15.1703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-designarenawebsite-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1229,"normalizedScore":74.0924,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aagpqadiamond-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.7,"normalizedScore":94.0379,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aahle-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":31.6,"normalizedScore":56.7729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aaomniscienceindex-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-18.5,"normalizedScore":53.9246,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-omniscienceaccuracy-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.5,"normalizedScore":48.6254,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-omnisciencehallucinationrate-2026-07-21","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73,"normalizedScore":28.9505,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-colbert-350m-nanobeirmultilingual-2026-07-21","modelSlug":"lfm2-5-colbert-350m","modelName":"LFM2.5-ColBERT-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-nanobeirmultilingual","benchmarkName":"NanoBEIR Multilingual Extended","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":60.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-colbert-350m-mkqa11-2026-07-21","modelSlug":"lfm2-5-colbert-350m","modelName":"LFM2.5-ColBERT-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mkqa11","benchmarkName":"MKQA-11 multilingual retrieval","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":69.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-embedding-350m-nanobeirmultilingual-2026-07-21","modelSlug":"lfm2-5-embedding-350m","modelName":"LFM2.5-Embedding-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-nanobeirmultilingual","benchmarkName":"NanoBEIR Multilingual Extended","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":57.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-embedding-350m-mkqa11-2026-07-21","modelSlug":"lfm2-5-embedding-350m","modelName":"LFM2.5-Embedding-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mkqa11","benchmarkName":"MKQA-11 multilingual retrieval","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":69.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-build-0-1-gertlabs-2026-07-21","modelSlug":"grok-build-0-1","modelName":"Grok Build 0.1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":49.15,"normalizedScore":49.6619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Build 0.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-tau2bench-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":20.8,"normalizedScore":20.9889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aascicode-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":20.9,"normalizedScore":33.7268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-lcr-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15,"normalizedScore":19.8151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-critpt-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aammmupro-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":44.6,"normalizedScore":31.3149,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-gpqa-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":43.4,"normalizedScore":25.667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-mmlupro-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":60,"normalizedScore":57.8828,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aagpqadiamond-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":37.5,"normalizedScore":23.3062,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aahle-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.8,"normalizedScore":3.3865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aaomniscienceindex-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-24,"normalizedScore":49.6075,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-omniscienceaccuracy-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.7,"normalizedScore":6.0137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.9,"normalizedScore":77.3221,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aaifbench-2026-07-21","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":35.1,"normalizedScore":27.6853,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-tau2bench-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":20.8,"normalizedScore":20.9889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aascicode-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":24.4,"normalizedScore":39.629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-lcr-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.7,"normalizedScore":40.5548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-critpt-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aammmupro-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":51.4,"normalizedScore":43.0796,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-gpqa-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":58.6,"normalizedScore":47.3534,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-mmlupro-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":69.4,"normalizedScore":71.2578,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aagpqadiamond-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":57.6,"normalizedScore":50.542,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aahle-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.7,"normalizedScore":1.1952,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aaomniscienceindex-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-20,"normalizedScore":52.7473,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-omniscienceaccuracy-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.6,"normalizedScore":9.2784,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-omnisciencehallucinationrate-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.3,"normalizedScore":79.2521,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aaifbench-2026-07-21","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":44.2,"normalizedScore":41.4523,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-tau2bench-2026-07-21","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":46.8,"normalizedScore":47.225,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aascicode-2026-07-21","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":26.4,"normalizedScore":43.0017,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-lcr-2026-07-21","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-critpt-2026-07-21","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aagpqadiamond-2026-07-21","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":73.8,"normalizedScore":72.4932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aahle-2026-07-21","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":10.1,"normalizedScore":13.9442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aaomniscienceindex-2026-07-21","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-59.5,"normalizedScore":21.7425,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-omniscienceaccuracy-2026-07-21","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.6,"normalizedScore":24.7423,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-omnisciencehallucinationrate-2026-07-21","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.5,"normalizedScore":4.222,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aaifbench-2026-07-21","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":34.4,"normalizedScore":26.6263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-tau2bench-2026-07-21","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":34.5,"normalizedScore":34.8133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aascicode-2026-07-21","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":19.2,"normalizedScore":30.86,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-lcr-2026-07-21","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-critpt-2026-07-21","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aagpqadiamond-2026-07-21","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":63.3,"normalizedScore":58.2656,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aahle-2026-07-21","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7,"normalizedScore":7.7689,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aaomniscienceindex-2026-07-21","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-72,"normalizedScore":11.9309,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-omniscienceaccuracy-2026-07-21","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":12.7,"normalizedScore":16.323,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-omnisciencehallucinationrate-2026-07-21","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":97,"normalizedScore":0,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aaifbench-2026-07-21","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":26.5,"normalizedScore":14.6747,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-tau2bench-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":24.3,"normalizedScore":24.5207,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aascicode-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.1,"normalizedScore":54.3002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-lcr-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28,"normalizedScore":36.9881,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-critpt-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aammmupro-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":53,"normalizedScore":45.8478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-designarenawebsite-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1109,"normalizedScore":54.2904,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aagpqadiamond-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":57.8,"normalizedScore":50.813,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aahle-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.3,"normalizedScore":2.3904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aaomniscienceindex-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-31.5,"normalizedScore":43.7206,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-omniscienceaccuracy-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.3,"normalizedScore":25.945,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-omnisciencehallucinationrate-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.9,"normalizedScore":43.5464,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aaifbench-2026-07-21","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.3,"normalizedScore":34.0393,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-tau2bench-2026-07-21","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":22.8,"normalizedScore":23.0071,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aascicode-2026-07-21","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":8.7,"normalizedScore":13.1535,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-lcr-2026-07-21","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4,"normalizedScore":5.284,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-critpt-2026-07-21","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aagpqadiamond-2026-07-21","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":28.1,"normalizedScore":10.5691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aahle-2026-07-21","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.1,"normalizedScore":3.9841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aaomniscienceindex-2026-07-21","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-81.8,"normalizedScore":4.2386,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-omniscienceaccuracy-2026-07-21","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.1,"normalizedScore":4.9828,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-omnisciencehallucinationrate-2026-07-21","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.5,"normalizedScore":4.222,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aaifbench-2026-07-21","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":20.5,"normalizedScore":5.5976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-tau2bench-2026-07-21","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":13.2,"normalizedScore":13.3199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aascicode-2026-07-21","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-lcr-2026-07-21","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-critpt-2026-07-21","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aagpqadiamond-2026-07-21","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":20.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aahle-2026-07-21","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.7,"normalizedScore":5.1793,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aaomniscienceindex-2026-07-21","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-72.1,"normalizedScore":11.8524,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-omniscienceaccuracy-2026-07-21","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-omnisciencehallucinationrate-2026-07-21","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.8,"normalizedScore":23.1604,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aaifbench-2026-07-21","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":16.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-tau2bench-2026-07-21","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":19.6,"normalizedScore":19.778,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aascicode-2026-07-21","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":8.2,"normalizedScore":12.3103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-lcr-2026-07-21","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.3,"normalizedScore":8.3223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-critpt-2026-07-21","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aagpqadiamond-2026-07-21","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":26.3,"normalizedScore":8.1301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aahle-2026-07-21","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5,"normalizedScore":3.7849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aaomniscienceindex-2026-07-21","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-73.6,"normalizedScore":10.675,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-omniscienceaccuracy-2026-07-21","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.3,"normalizedScore":3.6082,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-omnisciencehallucinationrate-2026-07-21","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.4,"normalizedScore":16.4053,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aaifbench-2026-07-21","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":26.2,"normalizedScore":14.2209,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-tau2bench-2026-07-21","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":14.6,"normalizedScore":14.7326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aascicode-2026-07-21","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":1.7,"normalizedScore":1.3491,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-lcr-2026-07-21","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-critpt-2026-07-21","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aagpqadiamond-2026-07-21","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":25.7,"normalizedScore":7.3171,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aahle-2026-07-21","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.4,"normalizedScore":6.5737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aaomniscienceindex-2026-07-21","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-87.2,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-omniscienceaccuracy-2026-07-21","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.7,"normalizedScore":0.8591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-omnisciencehallucinationrate-2026-07-21","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94.4,"normalizedScore":3.1363,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aaifbench-2026-07-21","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":17.6,"normalizedScore":1.2103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aascicode-2026-07-21","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.6,"normalizedScore":61.8887,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-lcr-2026-07-21","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.7,"normalizedScore":12.8137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aagpqadiamond-2026-07-21","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":61.5,"normalizedScore":55.8266,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aahle-2026-07-21","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.5,"normalizedScore":4.7809,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aaifbench-2026-07-21","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":22.9,"normalizedScore":9.2284,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-pro-gpqa-2026-07-21","modelSlug":"o1-pro","modelName":"o1-pro","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":79,"normalizedScore":76.4588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1-pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-tau2bench-2026-07-21","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":20.5,"normalizedScore":20.6862,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aascicode-2026-07-21","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":7.4,"normalizedScore":10.9612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-lcr-2026-07-21","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-critpt-2026-07-21","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aagpqadiamond-2026-07-21","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":42.4,"normalizedScore":29.9458,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aahle-2026-07-21","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.8,"normalizedScore":5.3785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aaomniscienceindex-2026-07-21","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-82.6,"normalizedScore":3.6107,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-omniscienceaccuracy-2026-07-21","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.7,"normalizedScore":2.5773,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-omnisciencehallucinationrate-2026-07-21","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.5,"normalizedScore":6.6345,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aaifbench-2026-07-21","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":25.3,"normalizedScore":12.8593,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-tau2bench-2026-07-21","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74.3,"normalizedScore":74.9748,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aascicode-2026-07-21","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.6,"normalizedScore":58.516,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-lcr-2026-07-21","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.7,"normalizedScore":73.5799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-critpt-2026-07-21","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aagpqadiamond-2026-07-21","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":78.3,"normalizedScore":78.5908,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aahle-2026-07-21","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":13.1,"normalizedScore":19.9203,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aaomniscienceindex-2026-07-21","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-57.9,"normalizedScore":22.9984,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-omniscienceaccuracy-2026-07-21","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.5,"normalizedScore":22.8522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-omnisciencehallucinationrate-2026-07-21","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.1,"normalizedScore":9.5296,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aaifbench-2026-07-21","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":64.7,"normalizedScore":72.466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-tau2bench-2026-07-21","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":31.9,"normalizedScore":32.1897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aascicode-2026-07-21","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":24.8,"normalizedScore":40.3035,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-lcr-2026-07-21","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-critpt-2026-07-21","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aagpqadiamond-2026-07-21","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":48.5095,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aahle-2026-07-21","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.8,"normalizedScore":1.3944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aaomniscienceindex-2026-07-21","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-61.7,"normalizedScore":20.0157,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-omniscienceaccuracy-2026-07-21","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.6,"normalizedScore":21.3058,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-omnisciencehallucinationrate-2026-07-21","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.5,"normalizedScore":6.6345,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aaifbench-2026-07-21","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":33.7,"normalizedScore":25.5673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.5.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-21","publishedAt":"2026-07-21","sourceId":"benchlm-public-dataset-2026-07-21","sourceTitle":"BenchLM public datasets — 21 July 2026","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/embed","sourceDate":"2026-07-21","sourceCheckedAt":"2026-07-21","checkedAt":"2026-07-21","verificationStatus":"source-checked"},{"resultId":"evidence-2026-07-2054","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"frontier-bench","benchmarkName":"Frontier-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Frontier-Bench","benchmarkVersion":"v0.1","score":43.3,"normalizedScore":43.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Frontier-Bench v0.1; mini-SWE-agent harness; GKE backend; mean reward over 5 attempts per task; Opus 4.8 safety-classifier fallback. Peak table score as published.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2055","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":1861,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"GDPval-AA v2 Elo as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2056","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"3","score":30.2,"normalizedScore":30.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"ARC-AGI-3 novel problem-solving score as published in the Anthropic Claude Opus 5 launch table (high effort).","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2057","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":90.8,"normalizedScore":90.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BrowseComp agentic search evaluation as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2058","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":64.7,"normalizedScore":64.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Humanity's Last Exam with tools as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2059","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":56.3,"normalizedScore":56.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Humanity's Last Exam without tools as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2060","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":64.7,"normalizedScore":64.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Humanity's Last Exam with tools as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2061","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":"2.0","score":70.6,"normalizedScore":70.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"OSWorld 2.0 computer-use evaluation as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2062","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2.0","score":70.6,"normalizedScore":70.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"OSWorld 2.0 as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2063","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"v1.1","score":68.8,"normalizedScore":68.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSWE v1.1 agentic coding as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2064","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"1.1","score":53.4,"normalizedScore":53.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"FrontierCode v1.1 Main as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2065","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-automationbench","benchmarkName":"AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":26,"normalizedScore":26,"unit":"percent","scoreDirection":"higher","modelConfiguration":"AutomationBench (Zapier-style business workflows) as published in the Anthropic Claude Opus 5 launch table. Not the AA AutomationBench scale.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2066","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"harvey-lab-aa","benchmarkName":"Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Harvey","benchmarkVersion":null,"score":11.7,"normalizedScore":11.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Legal Agent Benchmark held-out split as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2067","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":59.8,"normalizedScore":59.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"HealthBench Professional as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2068","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"biomystery-bench","benchmarkName":"BioMysteryBench","benchmarkCategory":"research","benchmarkOrganisation":"Anthropic","benchmarkVersion":"hard","score":49.4,"normalizedScore":49.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BioMysteryBench hard split as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-07-2069","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"biomystery-bench","benchmarkName":"BioMysteryBench","benchmarkCategory":"research","benchmarkOrganisation":"Anthropic","benchmarkVersion":"human-solved","score":90.1,"normalizedScore":90.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"BioMysteryBench human-solved subset as published in the Anthropic Claude Opus 5 launch table.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-24","publishedAt":"2026-07-24","sourceId":"anthropic-claude-opus-5-launch","sourceTitle":"Introducing Claude Opus 5","sourcePublisher":"Anthropic","sourceUrl":"https://www.anthropic.com/news/claude-opus-5","sourceDate":"2026-07-24","sourceCheckedAt":"2026-07-24","checkedAt":"2026-07-24","verificationStatus":"provider-reported"},{"resultId":"benchlm-ref-claude-mythos-5-terminalbench2-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":88,"normalizedScore":93.0605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-osworldverified-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":85,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-browsecomp-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":88,"normalizedScore":91.2134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-exploitgym-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":17.5,"normalizedScore":50.7599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-sweverified-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":95.5,"normalizedScore":99.3094,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-swepro-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":80.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-swemultimodal-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultimodal","benchmarkName":"SWE-bench Multimodal","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":54.9,"normalizedScore":85.623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-charxiv-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":93.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-charxivnotools-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":88.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-gpqa-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":94.1,"normalizedScore":98.0026,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-hle-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":64.5,"normalizedScore":99.6491,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-hlenotools-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":59,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-swemultilingual-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":92.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-usamo2026-2026-07-27","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":97.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-frontierbench-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontierbench","benchmarkName":"FrontierBench v0.1","benchmarkCategory":"agents","benchmarkOrganisation":"Ryan Marten, Alex Shaw, Andy Konwinski, Harbor, and the Laude Institute","benchmarkVersion":"2026","score":43.3,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-browsecomp-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":90.8,"normalizedScore":97.0711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-hlewithtools-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":64.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-deepsearchqa-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-draco-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-draco","benchmarkName":"Data Research and Analysis with Complex Operations","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":88.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-multiagentbrowsecompprerelease-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-multiagentbrowsecompprerelease","benchmarkName":"Multi-Agent BrowseComp — 10-agent team prerelease configuration","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":93.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-osworld2-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":70.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-mcpatlas-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":85.8,"normalizedScore":96.0751,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-mcpatlasclaimcoverage-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mcpatlasclaimcoverage","benchmarkName":"MCP-Atlas mean claim coverage","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":89.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-legalagentbenchallpass-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-legalagentbenchallpass","benchmarkName":"Legal Agent Benchmark all-pass rate — Anthropic harness","benchmarkCategory":"agents","benchmarkOrganisation":"Harvey AI and Anthropic","benchmarkVersion":"2026","score":23.58,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-legalagentbenchcriterionpass-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-legalagentbenchcriterionpass","benchmarkName":"Legal Agent Benchmark mean criterion-pass rate — Anthropic harness","benchmarkCategory":"agents","benchmarkOrganisation":"Harvey AI and Anthropic","benchmarkVersion":"2026","score":93.74,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-legalagentbenchheldoutallpass-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-legalagentbenchheldoutallpass","benchmarkName":"Legal Agent Benchmark all-pass rate — Harvey held-out set","benchmarkCategory":"agents","benchmarkOrganisation":"Harvey AI","benchmarkVersion":"2026","score":11.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-legalagentbenchheldoutcriterionpass-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-legalagentbenchheldoutcriterionpass","benchmarkName":"Legal Agent Benchmark mean criterion-pass rate — Harvey held-out set","benchmarkCategory":"agents","benchmarkOrganisation":"Harvey AI","benchmarkVersion":"2026","score":94.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-gdpvalaa-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1861,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-toolathlonverified-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-toolathlonverified","benchmarkName":"Toolathlon-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":80.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-toolathlonverifiedpass3-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-toolathlonverifiedpass3","benchmarkName":"Toolathlon Verified Pass@3","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":87,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-toolathlonverifiedpass3all-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-toolathlonverifiedpass3all","benchmarkName":"Toolathlon Verified Pass cubed","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":73.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-toolathlonverifiedavgturns-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-toolathlonverifiedavgturns","benchmarkName":"Toolathlon Verified average assistant turns","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":23.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-automationbench-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-automationbench","benchmarkName":"AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":26,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aaagenticindex-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.26,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-gdpvalaanormalized-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aatau3banking-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.3,"normalizedScore":81.6568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aabriefcaseelo-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1720,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aaterminalbench21-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-sweverified-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":96,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-swepro-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":79.2,"normalizedScore":97.0588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-swemultilingual-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":89.5,"normalizedScore":93.2836,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-swemultimodal-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultimodal","benchmarkName":"SWE-bench Multimodal","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":59.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-deepswe-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":68.8,"normalizedScore":87.9257,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-frontiercode-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":53.4,"normalizedScore":99.6575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-frontiercode11extended-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode11extended","benchmarkName":"FrontierCode 1.1 Extended","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":63.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-programbenchepisode1-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-programbenchepisode1","benchmarkName":"ProgramBench hidden-test pass rate after episode 1","benchmarkCategory":"coding","benchmarkOrganisation":"Yang et al.","benchmarkVersion":"2026","score":83,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-programbench-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-programbench","benchmarkName":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","benchmarkCategory":"coding","benchmarkOrganisation":"John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","benchmarkVersion":"2026","score":93,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aacodingindex-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.98,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aascicode-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":55.7,"normalizedScore":92.4115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-arcagi1-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-arcagi1","benchmarkName":"ARC-AGI-1 Semi-Private Evaluation","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":97.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-arcagi2-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":90.4,"normalizedScore":97.3384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-arcagi3-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":30.16,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-lcr-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70,"normalizedScore":92.4703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-critpt-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":29.1,"normalizedScore":90.0929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-chartography-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-chartography","benchmarkName":"Chartography without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":29.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-chartographywithtools-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-chartographywithtools","benchmarkName":"Chartography with image and code tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":83,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-benchcadvision2code-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-benchcadvision2code","benchmarkName":"BenchCAD Vision2Code voxel IoU without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Zhang et al. and Anthropic","benchmarkVersion":"2026","score":0.366,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-benchcadvision2codewithtools-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-benchcadvision2codewithtools","benchmarkName":"BenchCAD Vision2Code voxel IoU with tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Zhang et al. and Anthropic","benchmarkVersion":"2026","score":0.821,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-gdppdf-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdppdf","benchmarkName":"GDP.pdf mean criteria pass rate without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":83.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-gdppdfwithtools-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdppdfwithtools","benchmarkName":"GDP.pdf mean criteria pass rate with tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":85.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-officeqa-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-officeqa","benchmarkName":"OfficeQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Databricks and Anthropic","benchmarkVersion":"2026","score":78.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-officeqapro-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":66.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aammmupro-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":84.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-hle-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":64.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-hlenotools-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":56.3,"normalizedScore":94.9814,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-healthbench-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-healthbench","benchmarkName":"HealthBench raw score","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":67.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-healthbenchlengthadjusted-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-healthbenchlengthadjusted","benchmarkName":"HealthBench length-adjusted score","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":57.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-healthbenchprofessional-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":59.8,"normalizedScore":94.3548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-healthbenchprofessionalraw-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-healthbenchprofessionalraw","benchmarkName":"HealthBench Professional raw score","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":73.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-biomysterybenchhumansolvable-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-biomysterybenchhumansolvable","benchmarkName":"BioMysteryBench Human Solvable","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":90.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-biomysterybenchhumandifficult-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-biomysterybenchhumandifficult","benchmarkName":"BioMysteryBench Human Difficult","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":49.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-spatialbenchverified-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-spatialbenchverified","benchmarkName":"LatchBio SpatialBench Verified","benchmarkCategory":"knowledge","benchmarkOrganisation":"LatchBio and Anthropic","benchmarkVersion":"2026","score":72.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-singlecellbench-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-singlecellbench","benchmarkName":"LatchBio SingleCellBench","benchmarkCategory":"knowledge","benchmarkOrganisation":"LatchBio and Anthropic","benchmarkVersion":"2026","score":60.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-proteingymhard-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-proteingymhard","benchmarkName":"ProteinGym Hard","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":47.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-proteindesign-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-proteindesign","benchmarkName":"Anthropic Protein Design evaluation","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":42.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-organicchemistryv2-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-organicchemistryv2","benchmarkName":"Anthropic Organic Chemistry V2 evaluation","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":61.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-protocolstroubleshooting-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-protocolstroubleshooting","benchmarkName":"Molecular Biology Protocols Troubleshooting","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":61.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-protocolsunderstanding-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-protocolsunderstanding","benchmarkName":"Benchling Molecular Biology Protocols Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Benchling and Anthropic","benchmarkVersion":"2026","score":78.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aagpqadiamond-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":93.2,"normalizedScore":98.7216,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aaomniscienceindex-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.3,"normalizedScore":93.0141,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-omniscienceaccuracy-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.2,"normalizedScore":87.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-omnisciencehallucinationrate-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.1,"normalizedScore":56.5742,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-gmmlu-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gmmlu","benchmarkName":"Global MMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"Singh et al.","benchmarkVersion":"2024","score":92.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-milu-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-milu","benchmarkName":"Multi-task Indic Language Understanding Benchmark","benchmarkCategory":"knowledge","benchmarkOrganisation":"Verma et al.","benchmarkVersion":"2024","score":92.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-include-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-include","benchmarkName":"INCLUDE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":89.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-imo2026-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-imo2026","benchmarkName":"International Mathematical Olympiad 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":42,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-riemannbench-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-riemannbench","benchmarkName":"RiemannBench without tools","benchmarkCategory":"mathematics","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":60,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-riemannbenchwithtools-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-riemannbenchwithtools","benchmarkName":"RiemannBench with tools","benchmarkCategory":"mathematics","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":79,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-arxivmathjune2026-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-arxivmathjune2026","benchmarkName":"ArXivMath June 2026 without tools","benchmarkCategory":"mathematics","benchmarkOrganisation":"MathArena and Anthropic","benchmarkVersion":"2026","score":90.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-arxivmathjune2026withtools-2026-07-27","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-arxivmathjune2026withtools","benchmarkName":"ArXivMath June 2026 with tools","benchmarkCategory":"mathematics","benchmarkOrganisation":"MathArena and Anthropic","benchmarkVersion":"2026","score":91.3,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-terminalbench2-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":91.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-browsecomp-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":92.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-osworld2-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":62.6,"normalizedScore":88.2006,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-cybergym-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":84.5,"normalizedScore":94.508,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-exploitgym-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":33.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-toolathlon-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":58,"normalizedScore":63.8604,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaagenticindex-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54,"normalizedScore":97.7087,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-tau2bench-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":85.1,"normalizedScore":85.8729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61.8,"normalizedScore":90.8824,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gdpvalaa-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1736,"normalizedScore":93.6869,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aabriefcaseelo-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1505,"normalizedScore":80.1661,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaitbench-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aatau3banking-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33,"normalizedScore":97.6331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaautomationbench-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.2,"normalizedScore":92.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaharveylab-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.2,"normalizedScore":79.2717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-terminalbenchhard-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":65.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaterminalbench21-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88,"normalizedScore":95.6175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaenterpriseopsgym-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.9,"normalizedScore":64.0351,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-swepro-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":64.6,"normalizedScore":58.0214,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-deepswe-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":72.7,"normalizedScore":100,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiercode11extended-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode11extended","benchmarkName":"FrontierCode 1.1 Extended","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":60.6,"normalizedScore":64.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-vulcanbench-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vulcanbench","benchmarkName":"VulcanBench v3","benchmarkCategory":"coding","benchmarkOrganisation":"VulcanBench contributors","benchmarkVersion":"2026","score":87,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aacodingindex-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.39,"normalizedScore":99.1661,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aascicode-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":93.086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-arcagi2-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":92.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-arcagi3-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":7.78,"normalizedScore":25.5737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-genebenchpro-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-genebenchpro","benchmarkName":"GeneBench-Pro","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":28.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-lcr-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.7,"normalizedScore":97.358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-critpt-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":32.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-mmmupro-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":83,"normalizedScore":64.5161,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-mmmupropython-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":84.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aammmupro-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":83.4,"normalizedScore":97.7663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gpqa-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":94.6,"normalizedScore":98.7159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gpqadiamond-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":94.6,"normalizedScore":98.7159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-healthbenchprofessional-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":60.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-healthbenchhard-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":33.1,"normalizedScore":65.3571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aahle-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":47.2,"normalizedScore":87.8486,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.7,"normalizedScore":85.4788,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.5,"normalizedScore":95.0172,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.8,"normalizedScore":9.8914,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaifbench-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.7,"normalizedScore":84.9558,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiermath-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":89,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":89,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiermathv2tier4-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":83,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-exploitbench-2026-07-27","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-exploitbench","benchmarkName":"ExploitBench v8-bench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Seunghyun Lee, David Brumley, Carnegie Mellon University","benchmarkVersion":"2026","score":73.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-terminalbench2-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":84.3,"normalizedScore":86.4769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-osworldverified-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":85,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-gdpvalaa-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1747,"normalizedScore":94.2424,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaagenticindex-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.81,"normalizedScore":95.5446,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-tau2bench-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-gdpvalaanormalized-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.3,"normalizedScore":91.6176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aabriefcaseelo-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1574,"normalizedScore":86.5314,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaautomationbench-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.6,"normalizedScore":79.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaenterpriseopsgym-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaharveylab-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.6,"normalizedScore":97.1989,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aatau3banking-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":60.9467,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-terminalbenchhard-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":62.9,"normalizedScore":92.665,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaterminalbench21-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.6,"normalizedScore":82.0717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-sweverified-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":95,"normalizedScore":98.6188,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-swepro-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":80,"normalizedScore":99.1979,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-frontiercode-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":53.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-vulcanbench-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vulcanbench","benchmarkName":"VulcanBench v3","benchmarkCategory":"coding","benchmarkOrganisation":"VulcanBench contributors","benchmarkVersion":"2026","score":87,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aacodingindex-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76.49,"normalizedScore":97.894,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aascicode-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":60.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-lcr-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70,"normalizedScore":92.4703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-critpt-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":28.6,"normalizedScore":88.5449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-blueprintbench2-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"blueprint-bench-2","benchmarkName":"Blueprint-Bench 2","benchmarkCategory":"multimodal","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":38.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-officeqapro-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":57.9,"normalizedScore":61.3734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-designarenawebsite-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1332,"normalizedScore":91.0891,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aagpqadiamond-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":92.6,"normalizedScore":97.8693,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aahle-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":53.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaomniscienceindex-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.2,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-omniscienceaccuracy-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-omnisciencehallucinationrate-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.9,"normalizedScore":50.7841,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaifbench-2026-07-27","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":63.5,"normalizedScore":71.3864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-terminalbench2-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":74.6,"normalizedScore":69.2171,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-browsecomp-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":84.3,"normalizedScore":83.4728,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-deepsearchqa-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":93.1,"normalizedScore":94.0994,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-osworldverified-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":83.4,"normalizedScore":96.5217,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-financeagentv2-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":53.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gdpvalaa-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1593,"normalizedScore":86.4646,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-mcpatlas-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":82.2,"normalizedScore":89.9317,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-toolathlon-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":59.9,"normalizedScore":67.7618,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gertlabs-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":72.97,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaagenticindex-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.18,"normalizedScore":85.3064,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-tau2bench-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.4,"normalizedScore":95.2573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gdpvalaanormalized-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.6,"normalizedScore":80.2941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-researchclawbench-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":21.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-osworld2-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":20.6,"normalizedScore":26.2537,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aabriefcaseelo-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1346,"normalizedScore":65.4982,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaautomationbench-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.5,"normalizedScore":79,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaenterpriseopsgym-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44,"normalizedScore":68.8596,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaharveylab-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.1,"normalizedScore":90.1961,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aatau3banking-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.6,"normalizedScore":65.6805,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-terminalbenchhard-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":58.3,"normalizedScore":81.4181,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaterminalbench21-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.6,"normalizedScore":82.0717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-sweverified-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":88.6,"normalizedScore":89.779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-swepro-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":69.2,"normalizedScore":70.3209,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-swemultilingual-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":84.4,"normalizedScore":80.597,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-swemultimodal-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultimodal","benchmarkName":"SWE-bench Multimodal","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":38.4,"normalizedScore":32.9073,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aacodingindex-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.25,"normalizedScore":94.7279,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aascicode-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.5,"normalizedScore":88.7015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-frontiercode-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":46.5,"normalizedScore":76.0274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-arcagi2-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":72.08,"normalizedScore":74.1191,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-arcagi3-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":1.52,"normalizedScore":4.7556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-lcr-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.7,"normalizedScore":89.432,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-critpt-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":20.9,"normalizedScore":64.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-officeqapro-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":66.2,"normalizedScore":96.9957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-screenspotpro-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":87.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-charxiv-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":89.9,"normalizedScore":91.1765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-charxivnotools-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":80.5,"normalizedScore":29.4118,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-designarenawebsite-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1270,"normalizedScore":80.8581,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gpqa-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gpqadiamond-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-hle-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":57.9,"normalizedScore":88.0702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-hlenotools-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":49.8,"normalizedScore":82.8996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaomniscienceindex-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.4,"normalizedScore":89.9529,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-omniscienceaccuracy-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.6,"normalizedScore":74.5704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-omnisciencehallucinationrate-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.9,"normalizedScore":73.7033,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-include-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-include","benchmarkName":"INCLUDE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.6,"normalizedScore":67.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaifbench-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":62.2,"normalizedScore":69.469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-usamo2026-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":96.7,"normalizedScore":92.4306,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-frontiermathv2tiers13-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":47.241,"normalizedScore":53.0798,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-frontiermathv2tier4-2026-07-27","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":31.25,"normalizedScore":37.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-terminalbench2-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":69.7,"normalizedScore":60.4982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-qwenclawbench-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":64.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-qwenwebbench-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1568,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-claweval-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":65.2,"normalizedScore":83.3799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-bfclv4-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mcpatlas-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":76.4,"normalizedScore":80.0341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-vitabench-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":47.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-hlewithtools-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":53.5,"normalizedScore":58.9744,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaagenticindex-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.59,"normalizedScore":55.1373,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-tau2bench-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.7,"normalizedScore":95.56,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gdpvalaanormalized-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.6,"normalizedScore":56.7647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gdpvalaa-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1271,"normalizedScore":70.202,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gertlabs-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":64.27,"normalizedScore":81.6145,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-researchclawbench-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18.7,"normalizedScore":72.4138,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aabriefcaseelo-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":912,"normalizedScore":25.4613,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaenterpriseopsgym-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45,"normalizedScore":73.2456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaitbench-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.5,"normalizedScore":72.9249,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-terminalbenchhard-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":50.8,"normalizedScore":63.0807,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaterminalbench21-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.5,"normalizedScore":41.8327,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaharveylab-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.4,"normalizedScore":68.6275,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-sweverified-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.4,"normalizedScore":78.453,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-swepro-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":60.6,"normalizedScore":47.3262,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-swemultilingual-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.3,"normalizedScore":65.4229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-nl2repo-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":47.2,"normalizedScore":92.1659,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-scicode-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":53.5,"normalizedScore":80.0604,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-livecodebench-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":91.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aacodingindex-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.97,"normalizedScore":83.0247,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aascicode-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":48.8,"normalizedScore":80.7757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mrcrv2-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":90.4,"normalizedScore":93.6255,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-critpt-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":13.4,"normalizedScore":41.4861,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-lcr-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69,"normalizedScore":91.1493,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-designarenawebsite-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1293,"normalizedScore":84.6535,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gpqa-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.4,"normalizedScore":95.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gpqadiamond-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.4,"normalizedScore":95.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-hle-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":41.4,"normalizedScore":59.1228,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmlupro-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":89.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmluredux-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":95,"normalizedScore":93.9714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-supergpqa-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":73.6,"normalizedScore":70.2199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmmlu-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90.3,"normalizedScore":92,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaomniscienceindex-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.1,"normalizedScore":79.5133,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-omniscienceaccuracy-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.1,"normalizedScore":46.2199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-omnisciencehallucinationrate-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.9,"normalizedScore":89.3848,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmluprox-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":87,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-nova63-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59,"normalizedScore":97.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-include-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-include","benchmarkName":"INCLUDE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":47.0588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-maxife-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-maxife","benchmarkName":"MAXIFE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":89.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-polymath-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-polymath","benchmarkName":"PolyMath","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-ifeval-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.3,"normalizedScore":97.9314,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-ifbench-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":79.1,"normalizedScore":87.3391,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaifbench-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":80.5,"normalizedScore":96.4602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-hmmtfeb2026-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":97.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-imoanswerbench-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-apex-2026-07-27","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":44.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-terminalbench2-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":87.4,"normalizedScore":91.9929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-browsecomp-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":87.5,"normalizedScore":90.1674,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-osworld2-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":50.2,"normalizedScore":69.9115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-cybergym-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":81.8,"normalizedScore":88.3295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-exploitgym-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":23.2,"normalizedScore":68.0851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-toolathlon-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":53.1,"normalizedScore":53.7988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaagenticindex-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.38,"normalizedScore":85.6701,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-tau2bench-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86.3,"normalizedScore":87.0838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.1,"normalizedScore":79.5588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gdpvalaa-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1583,"normalizedScore":85.9596,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaharveylab-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.2,"normalizedScore":73.6695,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaitbench-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51,"normalizedScore":89.7233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aatau3banking-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.8,"normalizedScore":90.5325,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaautomationbench-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.6,"normalizedScore":64.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-terminalbenchhard-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":57.6,"normalizedScore":79.7066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaterminalbench21-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88,"normalizedScore":95.6175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-apexagentsaa-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":38.9,"normalizedScore":82.3276,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-swepro-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":63.4,"normalizedScore":54.8128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-deepswe-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":69.6,"normalizedScore":90.4025,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiercode11extended-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode11extended","benchmarkName":"FrontierCode 1.1 Extended","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":55.8,"normalizedScore":8.2353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aacodingindex-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76.66,"normalizedScore":98.1343,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aascicode-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.9,"normalizedScore":89.3761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-arcagi2-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":83.9,"normalizedScore":89.1001,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-arcagi3-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.8,"normalizedScore":2.3612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-lcr-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-critpt-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":30,"normalizedScore":92.8793,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-mmmupro-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":80.7,"normalizedScore":57.0968,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-mmmupropython-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":82,"normalizedScore":82.7815,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aammmupro-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.7,"normalizedScore":93.1271,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gpqa-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.9,"normalizedScore":96.2905,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gpqadiamond-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.9,"normalizedScore":96.2905,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-healthbenchprofessional-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":57.7,"normalizedScore":77.4194,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-healthbenchhard-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":32.7,"normalizedScore":63.9286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aahle-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":41.8,"normalizedScore":77.0916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-0.2,"normalizedScore":68.2889,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.9,"normalizedScore":73.3677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.2,"normalizedScore":14.234,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaifbench-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":71.2,"normalizedScore":82.7434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiermath-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":84.9,"normalizedScore":90.9292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":84.9,"normalizedScore":95.3933,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiermathv2tier4-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":68.3,"normalizedScore":82.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-exploitbench-2026-07-27","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-exploitbench","benchmarkName":"ExploitBench v8-bench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Seunghyun Lee, David Brumley, Carnegie Mellon University","benchmarkVersion":"2026","score":52.9,"normalizedScore":50.3614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-terminalbench2-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":80.4,"normalizedScore":79.5374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-browsecomp-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":84.7,"normalizedScore":84.3096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-hlewithtools-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":57.4,"normalizedScore":73.2601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-osworldverified-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":81.2,"normalizedScore":91.7391,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-gdpvalaa-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1603,"normalizedScore":86.9697,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaagenticindex-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.69,"normalizedScore":84.4153,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-gdpvalaanormalized-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.2,"normalizedScore":81.1765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aabriefcaseelo-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1386,"normalizedScore":69.1882,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaautomationbench-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.2,"normalizedScore":32.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaenterpriseopsgym-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.7,"normalizedScore":71.9298,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaharveylab-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.1,"normalizedScore":87.395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aatau3banking-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.2,"normalizedScore":69.2308,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaterminalbench21-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.5,"normalizedScore":65.7371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-sweverified-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":85.2,"normalizedScore":85.0829,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-swepro-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":63.2,"normalizedScore":54.2781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-swemultilingual-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.3,"normalizedScore":65.4229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-swemultimodal-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultimodal","benchmarkName":"SWE-bench Multimodal","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":28.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-frontiercode-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":42.7,"normalizedScore":63.0137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aacodingindex-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.55,"normalizedScore":90.9117,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aascicode-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.6,"normalizedScore":88.8702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-lcr-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70.7,"normalizedScore":93.395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-critpt-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":16.9,"normalizedScore":52.322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-charxiv-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":88.3,"normalizedScore":87.2549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-charxivnotools-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":77,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aammmupro-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":77.3,"normalizedScore":87.2852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-designarenawebsite-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1314,"normalizedScore":88.1188,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-hle-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":57.4,"normalizedScore":87.193,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-hlenotools-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":43.2,"normalizedScore":70.632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aagpqadiamond-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":91.1,"normalizedScore":95.7386,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaomniscienceindex-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.3,"normalizedScore":80.4553,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-omniscienceaccuracy-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.3,"normalizedScore":60.3093,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-omnisciencehallucinationrate-2026-07-27","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.3,"normalizedScore":72.0145,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-terminalbench2-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":70.3,"normalizedScore":61.5658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-qwenclawbench-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":61.8,"normalizedScore":80,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-qwenwebbench-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1536,"normalizedScore":81.2865,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-claweval-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.7,"normalizedScore":79.8883,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-bfclv4-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":72.9,"normalizedScore":96.1089,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mcpatlas-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":73.2,"normalizedScore":74.5734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-vitabench-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":45.6,"normalizedScore":92.9012,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-deepplanning-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":62.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-osworldverified-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":73.3,"normalizedScore":74.5652,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-androidworld-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":81,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aaagenticindex-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.81,"normalizedScore":37.3522,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-apexagentsaa-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":22.4,"normalizedScore":46.7672,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-tau2bench-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93,"normalizedScore":93.8446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gdpvalaanormalized-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.1,"normalizedScore":32.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gdpvalaa-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":943,"normalizedScore":53.6364,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-osworld2-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":2.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-sweverified-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.7,"normalizedScore":74.7238,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-swepro-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.6,"normalizedScore":39.3048,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-swemultilingual-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":75.8,"normalizedScore":59.204,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-nl2repo-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":41.1,"normalizedScore":64.0553,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-scicode-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":51.3,"normalizedScore":73.4139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-livecodebench-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":89.6,"normalizedScore":96.2963,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aacodingindex-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.86,"normalizedScore":68.735,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aascicode-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.5,"normalizedScore":75.2108,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-critpt-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":9.1,"normalizedScore":28.1734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mrcrv2-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":91.7,"normalizedScore":96.2151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-lcr-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65,"normalizedScore":85.8653,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmmupro-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79,"normalizedScore":51.6129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mathvision-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":90.3,"normalizedScore":80,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-charxiv-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":85.9,"normalizedScore":81.3725,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-erqa-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":69.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-medxpertqamm-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":71,"normalizedScore":68.4049,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-screenspotpro-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":79,"normalizedScore":78.91,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-simplevqa-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":81.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmsearchplus-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmsearchplus","benchmarkName":"MMSearch-Plus","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":41.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-realworldqa-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-omnidocbench15-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnidocbench15","benchmarkName":"OmniDocBench 1.5","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":91.4,"normalizedScore":88.2353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-ocrbenchv2-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ocrbench-v2","benchmarkName":"OCRBench v2","benchmarkCategory":"multimodal","benchmarkOrganisation":"OCRBench authors","benchmarkVersion":"2025","score":70.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-odinw13-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-odinw13","benchmarkName":"ODINW13","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":51.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-videommewithsub-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-videommmu-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":85.4,"normalizedScore":43.5897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mlvuavg-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mlvuavg","benchmarkName":"MLVU mean average","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aammmupro-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.5,"normalizedScore":92.7835,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-designarenawebsite-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1288,"normalizedScore":83.8284,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gpqa-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.3,"normalizedScore":92.581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gpqadiamond-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":90.3,"normalizedScore":92.581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-hle-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":34.7,"normalizedScore":47.3684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmlupro-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":88.5,"normalizedScore":98.4348,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmluredux-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.5,"normalizedScore":92.0874,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-supergpqa-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":71.4,"normalizedScore":67.1584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmmlu-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":89,"normalizedScore":74.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aaomniscienceindex-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.4,"normalizedScore":70.3297,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-omniscienceaccuracy-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.2,"normalizedScore":32.646,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-omnisciencehallucinationrate-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.5,"normalizedScore":86.2485,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmluprox-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":85.4,"normalizedScore":78.9474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-nova63-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":58.8,"normalizedScore":92.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-include-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-include","benchmarkName":"INCLUDE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-maxife-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-maxife","benchmarkName":"MAXIFE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-polymath-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-polymath","benchmarkName":"PolyMath","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-ifeval-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.6,"normalizedScore":98.818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-ifbench-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":79.1,"normalizedScore":87.3391,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aaifbench-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":78,"normalizedScore":92.7729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-hmmtfeb2026-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.9,"normalizedScore":94.1127,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-imoanswerbench-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":86,"normalizedScore":92.6874,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-apex-2026-07-27","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":22.7,"normalizedScore":50.5669,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-terminalbench2-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":88.3,"normalizedScore":93.5943,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-browsecomp-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":91.2,"normalizedScore":97.9079,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-deepsearchqa-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-toolathlonverified-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-toolathlonverified","benchmarkName":"Toolathlon-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":73.2,"normalizedScore":76.0518,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mcpatlas-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":84.2,"normalizedScore":93.3447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-automationbench-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-automationbench","benchmarkName":"AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":30.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-jobbench-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":52.9,"normalizedScore":96.1014,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-apexagents-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-apexagents","benchmarkName":"APEX-Agents","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI / APEX-Agents benchmark authors","benchmarkVersion":"2026","score":37.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-spreadsheetbench2-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-spreadsheetbench2","benchmarkName":"SpreadsheetBench 2","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":34.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-deckbench-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deckbench","benchmarkName":"DECK-Bench (Internal)","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":73.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaagenticindex-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.07,"normalizedScore":90.5619,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gdpvalaanormalized-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59.3,"normalizedScore":87.2059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gdpvalaa-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1686,"normalizedScore":91.1616,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aabriefcaseelo-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1541,"normalizedScore":83.4871,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaautomationbench-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaenterpriseopsgym-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.3,"normalizedScore":74.5614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaharveylab-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aatau3banking-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaterminalbench21-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85,"normalizedScore":83.6653,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-apexagentsaa-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":41.3,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaitbench-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.7,"normalizedScore":83.2016,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-deepswe-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":67.5,"normalizedScore":83.9009,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-frontierswe-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"frontierswe","benchmarkName":"FrontierSWE","benchmarkCategory":"coding","benchmarkOrganisation":"FrontierSWE","benchmarkVersion":"2026","score":81.2,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-programbench-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-programbench","benchmarkName":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","benchmarkCategory":"coding","benchmarkOrganisation":"John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","benchmarkVersion":"2026","score":77.8,"normalizedScore":61.4213,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-kimicodebenchv2-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"kimi-code-bench-v2","benchmarkName":"Kimi Code Bench v2","benchmarkCategory":"coding","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":72.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-swemarathon-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-marathon","benchmarkName":"SWE Marathon","benchmarkCategory":"coding","benchmarkOrganisation":"Abundant AI","benchmarkVersion":"2026","score":42,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-posttrainbench-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"posttrain-bench","benchmarkName":"PostTrainBench","benchmarkCategory":"coding","benchmarkOrganisation":"PostTrainBench","benchmarkVersion":"2026","score":36.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mlsbenchlite-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mlsbenchlite","benchmarkName":"MLS-Bench Lite","benchmarkCategory":"coding","benchmarkOrganisation":"MLS-Bench","benchmarkVersion":"2026","score":48.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aacodingindex-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76.24,"normalizedScore":97.5406,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aascicode-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":58.7,"normalizedScore":97.4705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-lcr-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.7,"normalizedScore":98.679,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-critpt-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":23.4,"normalizedScore":72.4458,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-officeqapro-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":63.3,"normalizedScore":84.5494,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mmmupro-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81.6,"normalizedScore":60,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mmmupropython-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.4,"normalizedScore":92.053,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-charxivnotools-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":84.8,"normalizedScore":65.5462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-charxiv-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":91.3,"normalizedScore":94.6078,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mathvision-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mathvisionpython-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mathvisionpython","benchmarkName":"MathVision with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI / MathVision authors","benchmarkVersion":"2026","score":97.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-babyvisionpython-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-babyvisionpython","benchmarkName":"BabyVision with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":85.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-zerobench-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"2026","score":23,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-zerobenchpython-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-zerobenchpython","benchmarkName":"ZeroBench_main with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI / ZeroBench authors","benchmarkVersion":"2026","score":41,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-worldvqaforceanswer-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-worldvqaforceanswer","benchmarkName":"WorldVQA ForceAnswer","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI / WorldVQA authors","benchmarkVersion":"2026","score":51,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-omnidocbench-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnidocbench","benchmarkName":"OmniDocBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI / OmniDocBench authors","benchmarkVersion":"2026","score":91.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-perceptionbench-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-perceptionbench","benchmarkName":"PerceptionBench (Internal)","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":58.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aammmupro-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.5,"normalizedScore":92.7835,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-designarenawebsite-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1386,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gpqa-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":93.5,"normalizedScore":97.1465,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gpqadiamond-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":93.5,"normalizedScore":97.1465,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-hle-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":56,"normalizedScore":84.7368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-hlenotools-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":43.5,"normalizedScore":71.1896,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaomniscienceindex-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.4,"normalizedScore":82.8885,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-omniscienceaccuracy-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46,"normalizedScore":73.5395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-omnisciencehallucinationrate-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.9,"normalizedScore":55.6092,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-exploitbench-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-exploitbench","benchmarkName":"ExploitBench v8-bench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Seunghyun Lee, David Brumley, Carnegie Mellon University","benchmarkVersion":"2026","score":32,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-acecyberrangesolved-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-acecyberrangesolved","benchmarkName":"ACE Cyber Range Challenges Solved","benchmarkCategory":"knowledge","benchmarkOrganisation":"NIST CAISI and UK AISI","benchmarkVersion":"2026","score":0,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-lastonescyberrangesteps-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-lastonescyberrangesteps","benchmarkName":"The Last Ones Average Progress","benchmarkCategory":"knowledge","benchmarkOrganisation":"NIST CAISI and UK AISI","benchmarkVersion":"2026","score":17,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-lastonescyberrangecompletion-2026-07-27","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-lastonescyberrangecompletion","benchmarkName":"The Last Ones Cyber Range Completion Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"NIST CAISI and UK AISI","benchmarkVersion":"2026","score":10,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-terminalbench2-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":80,"normalizedScore":78.8256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-mcpatlas-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":88.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-toolathlon-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":75.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-osworldverified-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":80.8,"normalizedScore":90.8696,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-deepsearchqa-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":84.9,"normalizedScore":68.6335,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-cybergym-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":59,"normalizedScore":36.1556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-financeagentv2-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":57.2,"normalizedScore":83.3123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-deepswe-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":53.3,"normalizedScore":39.9381,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-osworld2-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":14.2,"normalizedScore":16.8142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-jobbench-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":54.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-cybench-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"cybench","benchmarkName":"Cybench","benchmarkCategory":"agents","benchmarkOrganisation":"Stanford / Cybench authors","benchmarkVersion":"2025","score":92.9,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-exploitgym-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":0.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaagenticindex-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.54,"normalizedScore":67.776,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-gdpvalaanormalized-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.8,"normalizedScore":64.4118,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-gdpvalaa-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1375,"normalizedScore":75.4545,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aabriefcaseelo-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":868,"normalizedScore":21.4022,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaautomationbench-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.8,"normalizedScore":50.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaharveylab-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.1,"normalizedScore":95.7983,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aatau3banking-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.2,"normalizedScore":51.4793,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaterminalbench21-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.9,"normalizedScore":55.3785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaenterpriseopsgym-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.2,"normalizedScore":82.8947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-swepro-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":61.5,"normalizedScore":49.7326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aacodingindex-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.34,"normalizedScore":90.6148,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aascicode-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":58.2,"normalizedScore":96.6273,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-mrcr1m-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":54.1,"normalizedScore":48.3304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-lcr-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.3,"normalizedScore":83.6196,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-critpt-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":15.1,"normalizedScore":46.7492,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-charxiv-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":88.4,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-babyvision-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-babyvision","benchmarkName":"BabyVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":76.3,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-designarenawebsite-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1299,"normalizedScore":85.6436,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-hle-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":62.1,"normalizedScore":95.4386,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-hlenotools-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":52.2,"normalizedScore":87.3606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-healthbenchprofessional-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":59.3,"normalizedScore":90.3226,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aagpqadiamond-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.8,"normalizedScore":93.892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaomniscienceindex-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18,"normalizedScore":82.5746,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-omniscienceaccuracy-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.6,"normalizedScore":64.2612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-omnisciencehallucinationrate-2026-07-27","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.1,"normalizedScore":71.0495,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-terminalbench2-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":82,"normalizedScore":82.3843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-cybergym-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":81.8,"normalizedScore":88.3295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-browsecomp-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":84.4,"normalizedScore":83.682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-osworldverified-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":78.7,"normalizedScore":86.3043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mcpatlas-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":75.3,"normalizedScore":78.157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-toolathlon-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":55.6,"normalizedScore":58.9322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-tau2bench-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.9,"normalizedScore":94.7528,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaagenticindex-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.87,"normalizedScore":81.1057,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-apexagentsaa-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":37.7,"normalizedScore":79.7414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.6,"normalizedScore":72.9412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gdpvalaa-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1491,"normalizedScore":81.3131,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gertlabs-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":72.93,"normalizedScore":99.9155,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-researchclawbench-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":17,"normalizedScore":52.8736,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-osworld2-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":13,"normalizedScore":15.0442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-jobbench-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":42.7,"normalizedScore":74.0091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-exploitgym-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":13.4,"normalizedScore":38.2979,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aabriefcaseelo-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1153,"normalizedScore":47.6937,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaautomationbench-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.1,"normalizedScore":47,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaenterpriseopsgym-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.6,"normalizedScore":80.2632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaharveylab-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.3,"normalizedScore":76.7507,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaitbench-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.8,"normalizedScore":79.4466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aatau3banking-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.3,"normalizedScore":87.574,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-terminalbenchhard-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":60.6,"normalizedScore":87.0416,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaterminalbench21-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.3,"normalizedScore":80.8765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-swepro-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":58.6,"normalizedScore":41.9786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-vibecodebench-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":69.847,"normalizedScore":98.3719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-reactnativeevals-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":84.7,"normalizedScore":54.5817,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aacodingindex-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.89,"normalizedScore":95.6325,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aascicode-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":93.086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiercode-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":43,"normalizedScore":64.0411,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mrcrv2-64-128-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mrcrv2-64-128","benchmarkName":"OpenAI MRCR v2 8-needle 64K-128K","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mrcrv2-128-256-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mrcrv2-128-256","benchmarkName":"OpenAI MRCR v2 8-needle 128K-256K","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":87.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-arcagi2-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":85,"normalizedScore":90.4943,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-arcagi3-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.43,"normalizedScore":1.1307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-lcr-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.3,"normalizedScore":98.1506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-critpt-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":27.1,"normalizedScore":83.9009,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mmmupro-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81.2,"normalizedScore":58.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mmmupropython-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.2,"normalizedScore":90.7285,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-officeqapro-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":54.1,"normalizedScore":45.0644,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aammmupro-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":79.9,"normalizedScore":91.7526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-designarenawebsite-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1282,"normalizedScore":82.8383,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gpqa-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gpqadiamond-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-hle-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":52.2,"normalizedScore":78.0702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-hlenotools-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":41.4,"normalizedScore":67.2862,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.1,"normalizedScore":84.2229,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.9,"normalizedScore":92.268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.5,"normalizedScore":13.8721,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaifbench-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.9,"normalizedScore":89.6755,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiermath-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":51.7,"normalizedScore":17.4779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":51.7,"normalizedScore":58.0899,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiermathv2tier4-2026-07-27","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":35.4,"normalizedScore":42.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-pro-claweval-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.8,"normalizedScore":73.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-deepsearchqa-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":69.7,"normalizedScore":21.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-tau2bench-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.6,"normalizedScore":96.4682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaagenticindex-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.4,"normalizedScore":38.4252,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-apexagentsaa-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":32,"normalizedScore":67.4569,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gdpvalaanormalized-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.2,"normalizedScore":34.1176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gdpvalaa-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":965,"normalizedScore":54.7475,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gertlabs-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":56.87,"normalizedScore":65.9763,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-researchclawbench-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":13.3,"normalizedScore":10.3448,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaautomationbench-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.5,"normalizedScore":24,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaenterpriseopsgym-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.2,"normalizedScore":60.9649,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaharveylab-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaitbench-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.3,"normalizedScore":48.8142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aatau3banking-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-terminalbenchhard-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":53.8,"normalizedScore":70.4156,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaterminalbench21-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.8,"normalizedScore":39.0438,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-livecodebenchpro-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":82.9,"normalizedScore":88.4028,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-reactnativeevals-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":78.9,"normalizedScore":31.4741,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-vibecodebench-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":32.034,"normalizedScore":45.1164,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aacodingindex-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.83,"normalizedScore":87.0671,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aascicode-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":58.9,"normalizedScore":97.8078,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-arcagi2-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":77.08,"normalizedScore":80.4563,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-arcagi3-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.42,"normalizedScore":1.0974,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-lcr-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.7,"normalizedScore":96.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-critpt-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":17.7,"normalizedScore":54.7988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-mmmupro-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":83.9,"normalizedScore":67.4194,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-charxiv-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":80.2,"normalizedScore":67.402,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-erqa-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":69.4,"normalizedScore":97.8022,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-simplevqa-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":72.4,"normalizedScore":63.6719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-screenspotpro-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":84.4,"normalizedScore":91.7062,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-zerobench-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"2026","score":29,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-medxpertqamm-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":81.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aammmupro-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":82.4,"normalizedScore":96.0481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-designarenawebsite-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1281,"normalizedScore":82.6733,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gpqadiamond-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":94.3,"normalizedScore":98.2879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-hlenotools-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":45.4,"normalizedScore":74.7212,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-healthbenchhard-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":20.6,"normalizedScore":20.7143,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-medxpertqatext-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":71.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aahle-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":44.7,"normalizedScore":82.8685,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaomniscienceindex-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.9,"normalizedScore":94.27,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-omniscienceaccuracy-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.3,"normalizedScore":89.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.9,"normalizedScore":56.8154,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaglobalmmlulite-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaifbench-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":77.1,"normalizedScore":91.4454,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-frontiermathv2tiers13-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":36.9,"normalizedScore":41.4607,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-frontiermathv2tier4-2026-07-27","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":16.7,"normalizedScore":20.1205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-glm-5-2-terminalbench2-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":81,"normalizedScore":80.605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-mcpatlas-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":76.8,"normalizedScore":80.7167,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-toolathlon-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":48.2,"normalizedScore":43.7372,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaagenticindex-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.06,"normalizedScore":77.8141,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-tau2bench-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":99.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gdpvalaanormalized-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.5,"normalizedScore":74.2647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gdpvalaa-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1510,"normalizedScore":82.2727,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-apexagentsaa-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":33.7,"normalizedScore":71.1207,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-researchclawbench-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":20.7,"normalizedScore":95.4023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aabriefcaseelo-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1254,"normalizedScore":57.0111,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaenterpriseopsgym-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.7,"normalizedScore":63.1579,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaharveylab-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91,"normalizedScore":89.916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaitbench-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.7,"normalizedScore":73.3202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aatau3banking-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":60.9467,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-terminalbenchhard-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":50.8,"normalizedScore":63.0807,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaterminalbench21-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.9,"normalizedScore":55.3785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-swepro-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":62.1,"normalizedScore":51.3369,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-nl2repo-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":48.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-programbench-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-programbench","benchmarkName":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","benchmarkCategory":"coding","benchmarkOrganisation":"John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","benchmarkVersion":"2026","score":63.7,"normalizedScore":25.6345,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aacodingindex-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.76,"normalizedScore":86.9682,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aascicode-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50.5,"normalizedScore":83.6425,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-critpt-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":20.9,"normalizedScore":64.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-lcr-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.3,"normalizedScore":94.1876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-designarenawebsite-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1340,"normalizedScore":92.4092,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gpqa-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":91.2,"normalizedScore":93.865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gpqadiamond-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":91.2,"normalizedScore":93.865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hle-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":54.7,"normalizedScore":82.4561,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hlenotools-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":40.5,"normalizedScore":65.6134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaomniscienceindex-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4,"normalizedScore":71.5856,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-omniscienceaccuracy-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.1,"normalizedScore":37.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-omnisciencehallucinationrate-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.1,"normalizedScore":83.1122,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaopennessindex-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.4,"normalizedScore":22.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaifbench-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.3,"normalizedScore":85.8407,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aime2026-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":99.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hmmtnov2025-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":94.4,"normalizedScore":67.9487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hmmtfeb2026-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.5,"normalizedScore":93.552,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-mmanswerbench-2026-07-27","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":91,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-terminalbench2-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":84.7,"normalizedScore":87.1886,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-browsecomp-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.3,"normalizedScore":81.3808,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-osworld2-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":45.6,"normalizedScore":63.1268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-cybergym-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":77.9,"normalizedScore":79.405,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-exploitgym-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":12.4,"normalizedScore":35.2584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-toolathlon-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":53.4,"normalizedScore":54.4148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaagenticindex-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.6,"normalizedScore":82.4332,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.1,"normalizedScore":79.5588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gdpvalaa-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1582,"normalizedScore":85.9091,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaharveylab-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.9,"normalizedScore":81.2325,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaitbench-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.3,"normalizedScore":68.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aatau3banking-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.2,"normalizedScore":63.3136,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaautomationbench-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.2,"normalizedScore":47.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaterminalbench21-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.9,"normalizedScore":67.3307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-apexagentsaa-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":35.8,"normalizedScore":75.6466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-swepro-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":62.7,"normalizedScore":52.9412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-deepswe-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":67.2,"normalizedScore":82.9721,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiercode11extended-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode11extended","benchmarkName":"FrontierCode 1.1 Extended","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":55.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aacodingindex-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.45,"normalizedScore":90.7703,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aascicode-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":52.5,"normalizedScore":87.0152,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-arcagi2-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":59.54,"normalizedScore":58.2256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-arcagi3-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.18,"normalizedScore":0.2993,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-lcr-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-critpt-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":20.6,"normalizedScore":63.7771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-mmmupro-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.4,"normalizedScore":49.6774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-mmmupropython-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":79.5,"normalizedScore":66.2252,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aammmupro-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.6,"normalizedScore":89.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gpqa-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.3,"normalizedScore":95.4344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gpqadiamond-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.3,"normalizedScore":95.4344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-healthbenchprofessional-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":55.7,"normalizedScore":61.2903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-healthbenchhard-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":32,"normalizedScore":61.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aahle-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":37.2,"normalizedScore":67.9283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-11.2,"normalizedScore":59.6546,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.5,"normalizedScore":65.8076,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.1,"normalizedScore":8.3233,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiermath-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":78.6,"normalizedScore":76.9912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":78.6,"normalizedScore":88.3146,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiermathv2tier4-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":58.5,"normalizedScore":70.4819,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-exploitbench-2026-07-27","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-exploitbench","benchmarkName":"ExploitBench v8-bench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Seunghyun Lee, David Brumley, Carnegie Mellon University","benchmarkVersion":"2026","score":33.2,"normalizedScore":2.8916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-terminalbench2-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":76.2,"normalizedScore":72.0641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mcpatlas-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":83.6,"normalizedScore":92.3208,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-toolathlon-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":56.5,"normalizedScore":60.7803,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-osworldverified-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":78.4,"normalizedScore":85.6522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-financeagentv2-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":57.861,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gdpvalaa-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1344,"normalizedScore":73.8889,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-tau2bench-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.3,"normalizedScore":96.1655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gdpvalaanormalized-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.2,"normalizedScore":62.0588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaagenticindex-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.45,"normalizedScore":67.6123,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-apexagentsaa-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":47.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gertlabs-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":61.85,"normalizedScore":76.5004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-researchclawbench-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18,"normalizedScore":64.3678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaautomationbench-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.6,"normalizedScore":49.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaenterpriseopsgym-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.1,"normalizedScore":95.614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-swepro-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.1,"normalizedScore":32.6203,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-scicode-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":53.1,"normalizedScore":78.852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-vibecodebench-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":48.683,"normalizedScore":68.5647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aacodingindex-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70.14,"normalizedScore":88.9187,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aascicode-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.1,"normalizedScore":88.027,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mrcrv2-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":77.3,"normalizedScore":67.5299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mrcr1m-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":26.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-arcagi2-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":72.1,"normalizedScore":74.1445,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lcr-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":91.5456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-critpt-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":13.1,"normalizedScore":40.5573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-charxiv-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":84.2,"normalizedScore":77.2059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mmmupro-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":83.6,"normalizedScore":66.4516,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-blueprintbench2-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"blueprint-bench-2","benchmarkName":"Blueprint-Bench 2","benchmarkCategory":"multimodal","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":33.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aammmupro-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":84.3,"normalizedScore":99.3127,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-designarenawebsite-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1285,"normalizedScore":83.3333,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gpqa-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.2,"normalizedScore":95.2918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gpqadiamond-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.676,"normalizedScore":95.9709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-hle-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":40.2,"normalizedScore":57.0175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-omniscienceaccuracy-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.9,"normalizedScore":83.677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.7,"normalizedScore":43.7877,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaomniscienceindex-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.7,"normalizedScore":86.2637,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-ifbench-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":76.3,"normalizedScore":81.3305,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaifbench-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.3,"normalizedScore":90.2655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-frontiermathv2tiers13-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":38.966,"normalizedScore":43.782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-frontiermathv2tier4-2026-07-27","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.583,"normalizedScore":17.5699,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-terminalbench2-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":65.4,"normalizedScore":52.847,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-browsecomp-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.7,"normalizedScore":82.2176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-osworldverified-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":72.7,"normalizedScore":73.2609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-tau2bench-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-claweval-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":70.4,"normalizedScore":90.6425,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-deepsearchqa-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":73.7,"normalizedScore":33.8509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-cybergym-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":66.6,"normalizedScore":53.5469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-gertlabs-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":61.85,"normalizedScore":76.5004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-researchclawbench-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":19.9,"normalizedScore":86.2069,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-jobbench-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":36.7,"normalizedScore":61.0136,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-sweverified-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.84,"normalizedScore":79.0608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-sweverifiedarcee-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-livecodebenchpro-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":70.7,"normalizedScore":70.4932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-swepro-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":53.4,"normalizedScore":28.0749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-swerebench-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":65.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-reactnativeevals-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":84.1,"normalizedScore":52.1912,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-vibecodebench-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":57.573,"normalizedScore":81.0853,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aascicode-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.7,"normalizedScore":75.5481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-frontiercode-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":26.9,"normalizedScore":8.9041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-lcr-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.3,"normalizedScore":77.0145,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-critpt-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.8,"normalizedScore":8.6687,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-mmmupro-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":77.3,"normalizedScore":46.129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-erqa-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":51.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-screenspotpro-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":83.1,"normalizedScore":88.6256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-medxpertqamm-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":64.8,"normalizedScore":49.3865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aammmupro-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.5,"normalizedScore":79.0378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-designarenawebsite-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1325,"normalizedScore":89.934,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-gpqa-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":91.3,"normalizedScore":94.0077,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-gpqadiamond-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":89.2,"normalizedScore":91.0116,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-supergpqa-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-mmlupro-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":82,"normalizedScore":89.1861,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-mmluproarcee-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":89.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-hle-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":53,"normalizedScore":79.4737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-hlenotools-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":40,"normalizedScore":64.684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-healthbenchhard-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":14.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-medxpertqatext-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":52.1,"normalizedScore":8.9202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aaomniscienceindex-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.5,"normalizedScore":71.1931,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-omniscienceaccuracy-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.2,"normalizedScore":72.1649,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-omnisciencehallucinationrate-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76,"normalizedScore":25.3317,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aaifbench-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":44.6,"normalizedScore":43.5103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aime2025arcee-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":99.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-frontiermathv2tiers13-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":40.7,"normalizedScore":45.7303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-frontiermathv2tier4-2026-07-27","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":22.9,"normalizedScore":27.5904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-terminalbench2-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":75.1,"normalizedScore":70.1068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-cybergym-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":79,"normalizedScore":81.9222,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-browsecomp-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":82.7,"normalizedScore":80.1255,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-osworldverified-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":75,"normalizedScore":78.2609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mcpatlas-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":70.6,"normalizedScore":70.1365,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-toolathlon-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":54.6,"normalizedScore":56.8789,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-tau2bench-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":87.1,"normalizedScore":87.891,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-claweval-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":60.3,"normalizedScore":76.5363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-deepsearchqa-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":73.6,"normalizedScore":33.5404,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aaagenticindex-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.08,"normalizedScore":74.2135,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-apexagentsaa-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":33.3,"normalizedScore":70.2586,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.6,"normalizedScore":65.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gdpvalaa-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1392,"normalizedScore":76.3131,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gertlabs-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":64.89,"normalizedScore":82.9248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-researchclawbench-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":15.3,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-jobbench-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":38.9,"normalizedScore":65.7786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-exploitgym-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":6,"normalizedScore":15.8055,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-livecodebenchpro-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":87.5,"normalizedScore":95.1556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-swepro-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.7,"normalizedScore":39.5722,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-reactnativeevals-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":85.3,"normalizedScore":56.9721,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-vibecodebench-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":67.421,"normalizedScore":94.9551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aacodingindex-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.05,"normalizedScore":90.2049,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aascicode-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.6,"normalizedScore":93.9292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-arcagi2-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":73.95,"normalizedScore":76.4892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-arcagi3-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.21,"normalizedScore":0.3991,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-lcr-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-critpt-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":23.4,"normalizedScore":72.4458,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mmmupro-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81.2,"normalizedScore":58.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-officeqapro-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":53.2,"normalizedScore":41.2017,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mmmupropython-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":82.1,"normalizedScore":83.4437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-charxiv-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":82.8,"normalizedScore":73.7745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-erqa-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":65.4,"normalizedScore":75.8242,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-simplevqa-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":61.1,"normalizedScore":19.5313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-screenspotpro-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":85.4,"normalizedScore":94.0758,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-zerobench-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"2026","score":41,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-medxpertqamm-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":77.1,"normalizedScore":87.1166,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aammmupro-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.4,"normalizedScore":89.1753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-designarenawebsite-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1250,"normalizedScore":77.5578,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gpqa-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.8,"normalizedScore":96.1478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-hle-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":52.1,"normalizedScore":77.8947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-hlenotools-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":39.8,"normalizedScore":64.3123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gpqadiamond-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.8,"normalizedScore":96.1478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-healthbenchhard-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":40.1,"normalizedScore":90.3571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-medxpertqatext-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":59.6,"normalizedScore":44.1315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.7,"normalizedScore":72.9199,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50,"normalizedScore":80.4124,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.6,"normalizedScore":10.1327,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-healthbenchprofessional-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":48.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aaifbench-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.9,"normalizedScore":86.7257,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":47.6,"normalizedScore":53.4831,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-frontiermathv2tier4-2026-07-27","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":27.1,"normalizedScore":32.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-terminalbench2-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":67.9,"normalizedScore":57.2954,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-browsecomp-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.4,"normalizedScore":81.59,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-hlewithtools-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":48.2,"normalizedScore":39.5604,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-mcpatlas-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":73.6,"normalizedScore":75.256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gdpvalaa-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1306,"normalizedScore":71.9697,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-toolathlon-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":51.8,"normalizedScore":51.1294,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaagenticindex-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":36.36,"normalizedScore":65.6301,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-apexagentsaa-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":24.3,"normalizedScore":50.8621,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-tau2bench-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":96.2,"normalizedScore":97.0737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gdpvalaanormalized-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.3,"normalizedScore":59.2647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aabriefcaseelo-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":931,"normalizedScore":27.214,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaenterpriseopsgym-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.4,"normalizedScore":53.0702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaharveylab-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.4,"normalizedScore":71.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaitbench-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.3,"normalizedScore":64.6245,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aatau3banking-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.8,"normalizedScore":55.0296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-terminalbenchhard-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":46.2,"normalizedScore":51.8337,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaterminalbench21-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-codeforces-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-codeforces","benchmarkName":"Codeforces Rating","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":3206,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-sweverified-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.6,"normalizedScore":78.7293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-swepro-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.4,"normalizedScore":33.4225,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-swemultilingual-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":76.2,"normalizedScore":60.199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-vibecodebench-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":49.931,"normalizedScore":70.3224,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aacodingindex-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59.36,"normalizedScore":73.682,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aascicode-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50,"normalizedScore":82.7993,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-mrcr1m-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":83.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-corpusqa1m-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":62,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-lcr-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.3,"normalizedScore":87.5826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-critpt-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":12.9,"normalizedScore":39.9381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-designarenawebsite-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1264,"normalizedScore":79.868,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-mmlupro-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":87.5,"normalizedScore":97.012,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-simpleqa-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":57.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-chinesesimpleqa-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":84.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gpqa-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.1,"normalizedScore":92.2956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gpqadiamond-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":90.1,"normalizedScore":92.2956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-hle-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":37.7,"normalizedScore":52.6316,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaomniscienceindex-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10,"normalizedScore":60.5965,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-omniscienceaccuracy-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.3,"normalizedScore":68.9003,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-omnisciencehallucinationrate-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94,"normalizedScore":3.6188,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaopennessindex-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50,"normalizedScore":33.4,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaifbench-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.5,"normalizedScore":90.5605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-hmmtfeb2026-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":95.2,"normalizedScore":97.3367,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-imoanswerbench-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":89.8,"normalizedScore":99.6344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-apex-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":38.3,"normalizedScore":85.941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-apexshortlist-2026-07-27","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-terminalbench2-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":66.7,"normalizedScore":55.1601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-browsecomp-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.2,"normalizedScore":81.1715,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-osworldverified-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":73.1,"normalizedScore":74.1304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-toolathlon-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":50,"normalizedScore":47.4333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mcpatlas-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":55.9,"normalizedScore":45.0512,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-claweval-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.3,"normalizedScore":79.3296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-deepsearchqa-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":92.5,"normalizedScore":92.236,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-wideresearch-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":80.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aaagenticindex-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.27,"normalizedScore":54.5554,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-tau2bench-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gdpvalaanormalized-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.4,"normalizedScore":50.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gdpvalaa-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1188,"normalizedScore":66.0101,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-apexagentsaa-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":28.5,"normalizedScore":59.9138,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gertlabs-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":56.82,"normalizedScore":65.8707,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-researchclawbench-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18,"normalizedScore":64.3678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-osworld2-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.6549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-sweverified-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.2,"normalizedScore":78.1768,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-livecodebenchv6-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":89.6,"normalizedScore":93.9678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-swepro-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":58.6,"normalizedScore":41.9786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-swemultilingual-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":76.7,"normalizedScore":61.4428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-scicode-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":52.2,"normalizedScore":76.1329,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-vibecodebench-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":37.891,"normalizedScore":53.3654,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aacodingindex-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61.77,"normalizedScore":77.0883,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aascicode-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.5,"normalizedScore":88.7015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-lcr-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-critpt-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":8,"normalizedScore":24.7678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mmmupro-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79.4,"normalizedScore":52.9032,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mmmupropython-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":80.1,"normalizedScore":70.1987,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-charxiv-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":80.4,"normalizedScore":67.8922,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mathvision-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.4,"normalizedScore":65.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-vstar-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":96.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aammmupro-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":79.4,"normalizedScore":90.8935,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-designarenawebsite-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1306,"normalizedScore":86.7987,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gpqa-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.5,"normalizedScore":92.8663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gpqadiamond-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":90.5,"normalizedScore":92.8663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-hle-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":34.7,"normalizedScore":47.3684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aaomniscienceindex-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.4,"normalizedScore":73.4694,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-omniscienceaccuracy-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.8,"normalizedScore":50.8591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-omnisciencehallucinationrate-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.3,"normalizedScore":69.6019,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aaifbench-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76,"normalizedScore":89.823,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aime2026-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":96.4,"normalizedScore":95.2365,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-hmmtfeb2026-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.7,"normalizedScore":93.8324,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mmanswerbench-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86,"normalizedScore":58.6777,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-frontiermathv2tiers13-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":38.966,"normalizedScore":43.782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-frontiermathv2tier4-2026-07-27","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.58,"normalizedScore":17.5663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-terminalbench2-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":63.5,"normalizedScore":49.4662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-browsecomp-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":68,"normalizedScore":49.3724,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-tau3bench-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.6,"normalizedScore":19.3798,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-mcpatlas-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":71.8,"normalizedScore":72.1843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-cybergym-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":68.7,"normalizedScore":58.3524,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-claweval-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.3,"normalizedScore":79.3296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aaagenticindex-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.87,"normalizedScore":53.828,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-tau2bench-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":97.7,"normalizedScore":98.5873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gdpvalaanormalized-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.8,"normalizedScore":55.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gertlabs-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":60.11,"normalizedScore":72.8233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gdpvalaa-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1256,"normalizedScore":69.4444,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-researchclawbench-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18.2,"normalizedScore":66.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-swepro-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":58.4,"normalizedScore":41.4439,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-nl2repo-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":42.7,"normalizedScore":71.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-swerebench-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":62.7,"normalizedScore":89.0295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-vibecodebench-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":31.456,"normalizedScore":44.3024,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aacodingindex-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.78,"normalizedScore":68.6219,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aascicode-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.8,"normalizedScore":72.344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-lcr-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.3,"normalizedScore":82.2985,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-critpt-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.6,"normalizedScore":14.2415,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-designarenawebsite-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1305,"normalizedScore":86.6337,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gpqadiamond-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":86.2,"normalizedScore":86.7313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-hle-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":52.3,"normalizedScore":78.2456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aaomniscienceindex-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.9,"normalizedScore":69.9372,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-omniscienceaccuracy-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.2,"normalizedScore":36.0825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-omnisciencehallucinationrate-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.4,"normalizedScore":81.544,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aaifbench-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.3,"normalizedScore":90.2655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aime2026-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.3,"normalizedScore":93.3651,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-hmmtnov2025-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":94,"normalizedScore":62.8205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-hmmtfeb2026-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":82.6,"normalizedScore":79.6748,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-mmanswerbench-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.8,"normalizedScore":40.4959,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-frontiermathv2tiers13-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":33.448,"normalizedScore":37.582,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-frontiermathv2tier4-2026-07-27","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":12.5,"normalizedScore":15.0602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-terminalbench2-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59.1,"normalizedScore":41.637,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-osworldverified-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":72.1,"normalizedScore":71.9565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-claweval-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":67.8,"normalizedScore":87.0112,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-cybergym-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":65.2,"normalizedScore":50.3432,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-tau2bench-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":79.5,"normalizedScore":80.222,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-gertlabs-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":62.92,"normalizedScore":78.7616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-osworld2-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":8.3,"normalizedScore":8.1121,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-jobbench-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":36.9,"normalizedScore":61.4468,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-sweverified-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":79.6,"normalizedScore":77.3481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-swerebench-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":60.7,"normalizedScore":80.5907,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-reactnativeevals-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":80.6,"normalizedScore":38.247,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-vibecodebench-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":51.476,"normalizedScore":72.4983,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aascicode-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.9,"normalizedScore":77.5717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-frontiercode-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":24.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-lcr-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":57.7,"normalizedScore":76.2219,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-critpt-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-charxiv-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":77.4,"normalizedScore":60.5392,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aammmupro-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":70.6,"normalizedScore":75.7732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-designarenawebsite-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1314,"normalizedScore":88.1188,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-gpqa-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":89.9,"normalizedScore":92.0103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-supergpqa-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-mmlupro-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":79.2,"normalizedScore":85.202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-hle-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":49,"normalizedScore":72.4561,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aagpqadiamond-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":79.9,"normalizedScore":79.8295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aaomniscienceindex-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-2.9,"normalizedScore":66.1695,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-omniscienceaccuracy-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38,"normalizedScore":59.7938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-omnisciencehallucinationrate-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.9,"normalizedScore":37.5151,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aaifbench-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":41.2,"normalizedScore":38.4956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-frontiermathv2tiers13-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":32.4,"normalizedScore":36.4045,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-frontiermathv2tier4-2026-07-27","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":8.3,"normalizedScore":10,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-terminalbench2-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":61.6,"normalizedScore":46.0854,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-claweval-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":58.8,"normalizedScore":74.4413,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-qwenclawbench-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":57.2,"normalizedScore":43.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-tau3bench-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.7,"normalizedScore":19.7674,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-vitabench-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":44.3,"normalizedScore":88.8889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-deepplanning-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":41.5,"normalizedScore":56.5762,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-toolathlon-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":39.8,"normalizedScore":26.4887,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mcpatlas-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":48.2,"normalizedScore":31.9113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mcptasks-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.1,"normalizedScore":99.3377,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-wideresearch-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.3,"normalizedScore":68.599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aaagenticindex-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.55,"normalizedScore":49.609,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-tau2bench-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":97.7,"normalizedScore":98.5873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gdpvalaanormalized-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.9,"normalizedScore":46.9118,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gdpvalaa-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1139,"normalizedScore":63.5354,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gertlabs-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":50.6,"normalizedScore":52.7261,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-researchclawbench-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18,"normalizedScore":64.3678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-sweverified-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":78.8,"normalizedScore":76.2431,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-swepro-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.6,"normalizedScore":36.631,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-swemultilingual-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.8,"normalizedScore":54.2289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-livecodebenchv6-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":87.1,"normalizedScore":89.7788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-vibecodebench-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":25.564,"normalizedScore":36.0041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aacodingindex-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.53,"normalizedScore":66.8551,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aascicode-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.7,"normalizedScore":67.1164,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aineedle-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":68.3,"normalizedScore":46.729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-longbenchv2-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":62,"normalizedScore":87.8173,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-lcr-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-critpt-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.9,"normalizedScore":8.9783,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmmu-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":86,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmmupro-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.8,"normalizedScore":50.9677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mathvision-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88,"normalizedScore":68.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-videommmu-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84,"normalizedScore":7.6923,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-screenspotpro-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":68.2,"normalizedScore":53.3175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-charxiv-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":81.5,"normalizedScore":70.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-vstar-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":96.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aammmupro-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78,"normalizedScore":88.488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-designarenawebsite-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1249,"normalizedScore":77.3927,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gpqa-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.4,"normalizedScore":92.7236,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-supergpqa-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":71.6,"normalizedScore":67.4367,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmlupro-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":88.5,"normalizedScore":98.4348,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmluredux-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.5,"normalizedScore":92.0874,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-ceval-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":93.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hle-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":28.8,"normalizedScore":37.0175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aagpqadiamond-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":88.2,"normalizedScore":91.6193,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aaomniscienceindex-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.7,"normalizedScore":70.5651,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-omniscienceaccuracy-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.2,"normalizedScore":39.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-omnisciencehallucinationrate-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32,"normalizedScore":78.4077,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmluprox-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":84.7,"normalizedScore":69.7368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-nova63-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":57.9,"normalizedScore":70,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-ifeval-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.3,"normalizedScore":97.9314,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-ifbench-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":75.8,"normalizedScore":80.2575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aaifbench-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.2,"normalizedScore":88.6431,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aime2026-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.3,"normalizedScore":93.3651,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hmmtfeb2025-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":96.7,"normalizedScore":88.2353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hmmtnov2025-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":94.6,"normalizedScore":70.5128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hmmtfeb2026-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.8,"normalizedScore":86.9638,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmanswerbench-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.8,"normalizedScore":40.4959,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-frontiermathv2tiers13-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":26.207,"normalizedScore":29.4461,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-frontiermathv2tier4-2026-07-27","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":8.333,"normalizedScore":10.0398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-terminalbench2-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":56.2,"normalizedScore":36.4769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-claweval-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.7,"normalizedScore":72.905,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-qwenclawbench-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":54.1,"normalizedScore":18.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-tau3bench-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":65.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-deepplanning-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":14.6,"normalizedScore":0.4175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-toolathlon-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":38,"normalizedScore":22.7926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mcpatlas-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":31.1,"normalizedScore":2.7304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mcptasks-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":60.8,"normalizedScore":11.2583,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-wideresearch-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":69.8,"normalizedScore":46.8599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-tau2bench-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.2,"normalizedScore":99.0918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-cybergym-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":43.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-apexagentsaa-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":14.5,"normalizedScore":29.7414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-gertlabs-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":50.99,"normalizedScore":53.5503,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-sweverified-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.8,"normalizedScore":74.8619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-sweverifiedarcee-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":72.8,"normalizedScore":77.4194,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-swepro-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.1,"normalizedScore":32.6203,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-swemultilingual-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.3,"normalizedScore":52.9851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-swerebench-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":62.8,"normalizedScore":89.4515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-reactnativeevals-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":74.8,"normalizedScore":15.1394,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aascicode-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.2,"normalizedScore":76.3912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-longbenchv2-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.8,"normalizedScore":81.7259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aineedle-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":63.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-lcr-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.3,"normalizedScore":83.6196,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-critpt-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2,"normalizedScore":6.192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-designarenawebsite-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1278,"normalizedScore":82.1782,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-gpqa-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":86,"normalizedScore":86.446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-gpqadiamond-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":86,"normalizedScore":86.446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-supergpqa-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":66.8,"normalizedScore":60.757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmlupro-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.7,"normalizedScore":94.4508,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmluproarcee-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":85.8,"normalizedScore":76.259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hle-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":50.4,"normalizedScore":74.9123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aaomniscienceindex-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2,"normalizedScore":70.0157,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-omniscienceaccuracy-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.9,"normalizedScore":40.7216,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-omnisciencehallucinationrate-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34,"normalizedScore":75.9952,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmluprox-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":83.1,"normalizedScore":48.6842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-nova63-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":55.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-ifeval-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":92.6,"normalizedScore":92.9078,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aaifbench-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.3,"normalizedScore":84.3658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aime2026-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.8,"normalizedScore":94.2157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aime2025arcee-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":93.3,"normalizedScore":91.4248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hmmtfeb2025-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":97.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hmmtnov2025-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":96.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hmmtfeb2026-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.4,"normalizedScore":85.0014,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmanswerbench-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":82.5,"normalizedScore":29.7521,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-frontiermathv2tiers13-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":16.434,"normalizedScore":18.4652,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-frontiermathv2tier4-2026-07-27","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.1,"normalizedScore":2.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-terminalbench2-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":63.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-browsecomp-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":77.1,"normalizedScore":68.41,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-mcpatlas-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":74.1,"normalizedScore":76.1092,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-designarenaagenticwebdev-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-designarenaagenticwebdev","benchmarkName":"Design Arena Agentic Web Dev Elo","benchmarkCategory":"agents","benchmarkOrganisation":"Design Arena / Intelligence","benchmarkVersion":"2026","score":1257,"normalizedScore":50,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aaagenticindex-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.34,"normalizedScore":58.3197,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gdpvalaanormalized-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":36.8,"normalizedScore":54.1176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gdpvalaa-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1237,"normalizedScore":68.4848,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aabriefcaseelo-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":839,"normalizedScore":18.7269,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aatau3banking-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.7,"normalizedScore":42.6036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aaenterpriseopsgym-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.1,"normalizedScore":42.9825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-sweverified-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.6,"normalizedScore":74.5856,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-swepro-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":54.3,"normalizedScore":30.4813,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aacodingindex-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.06,"normalizedScore":63.364,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aascicode-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.1,"normalizedScore":76.2226,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-lcr-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.3,"normalizedScore":83.6196,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-critpt-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.4,"normalizedScore":16.7183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-mmmupro-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":73.5,"normalizedScore":33.871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-charxiv-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":82,"normalizedScore":71.8137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-charxivnotools-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":78.1,"normalizedScore":9.2437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aammmupro-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":73.5,"normalizedScore":80.756,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gpqa-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.9,"normalizedScore":89.1568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gpqadiamond-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87.9,"normalizedScore":89.1568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-hle-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":46,"normalizedScore":67.193,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-hlenotools-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":30,"normalizedScore":46.0967,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aaomniscienceindex-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.1,"normalizedScore":70.0942,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-omniscienceaccuracy-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40,"normalizedScore":63.2302,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-omnisciencehallucinationrate-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.1,"normalizedScore":40.8926,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aaopennessindex-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-ifbench-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":79.8,"normalizedScore":88.8412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aime2026-2026-07-27","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":97.1,"normalizedScore":96.4274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-terminalbench2-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":66,"normalizedScore":53.9146,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-browsecomp-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.52,"normalizedScore":81.841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-osworldverified-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":70.06,"normalizedScore":67.5217,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-mcpatlas-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":74.2,"normalizedScore":76.2799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-claweval-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":74.5,"normalizedScore":96.3687,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaagenticindex-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.36,"normalizedScore":63.8116,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-tau2bench-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":88.9,"normalizedScore":89.7074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-gdpvalaanormalized-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.5,"normalizedScore":65.4412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-gdpvalaa-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1391,"normalizedScore":76.2626,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-gdpvalrubrics-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gdpvalrubrics","benchmarkName":"GDPval rubrics","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":74.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-bankertoolbench-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-bankertoolbench","benchmarkName":"BankerToolBench","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":76.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-researchclawbench-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":19.8,"normalizedScore":85.0575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-osworld2-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.6549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aabriefcaseelo-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1109,"normalizedScore":43.6347,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaenterpriseopsgym-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.1,"normalizedScore":16.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaharveylab-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.4,"normalizedScore":82.6331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-terminalbenchhard-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":42.4,"normalizedScore":42.5428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaterminalbench21-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.2,"normalizedScore":4.7809,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-sweverified-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.5,"normalizedScore":78.5912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-swepro-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":59,"normalizedScore":43.0481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-nl2repo-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":42.13,"normalizedScore":68.8018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aacodingindex-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.57,"normalizedScore":72.5654,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aascicode-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.4,"normalizedScore":75.0422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-vibev2-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibev2","benchmarkName":"VIBE V2","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":50.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-svgbench-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-svgbench","benchmarkName":"SVG-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":63.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-kernelbenchhard-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-kernelbenchhard","benchmarkName":"KernelBench Hard","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":28.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-lcr-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-critpt-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.7,"normalizedScore":11.4551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-officeqapro-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":45.1,"normalizedScore":6.4378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-omnidocbench15-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omnidocbench15","benchmarkName":"OmniDocBench 1.5","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":91.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-mmmupro-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.1,"normalizedScore":48.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-videommmu-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.6,"normalizedScore":23.0769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-videommewithsub-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-designarenawebsite-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1289,"normalizedScore":83.9934,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aammmupro-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.6,"normalizedScore":89.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aagpqadiamond-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":92.9,"normalizedScore":98.2955,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aahle-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":37.1,"normalizedScore":67.7291,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaomniscienceindex-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.4,"normalizedScore":69.5447,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-omniscienceaccuracy-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15,"normalizedScore":20.2749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-omnisciencehallucinationrate-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.1,"normalizedScore":97.5875,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaopennessindex-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaifbench-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":82.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-usamo2026-2026-07-27","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":85.71,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-terminalbench2-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59.3,"normalizedScore":41.9929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-osworldverified-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":66.3,"normalizedScore":59.3478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-osworld-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":"2026","score":66.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-claweval-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":59.6,"normalizedScore":75.5587,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-qwenclawbench-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":52.3,"normalizedScore":4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-tau3bench-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.2,"normalizedScore":17.8295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-vitabench-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":23.3,"normalizedScore":24.0741,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-deepplanning-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":26.4,"normalizedScore":25.0522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-toolathlon-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":43.5,"normalizedScore":34.0862,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mcpatlas-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":42.3,"normalizedScore":21.843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mcptasks-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":71.8,"normalizedScore":84.106,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-wideresearch-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":76.4,"normalizedScore":78.744,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-cybergym-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":50.6,"normalizedScore":16.9336,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-tau2bench-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86.3,"normalizedScore":87.0838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-gertlabs-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":64.23,"normalizedScore":81.53,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-jobbench-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":32.3,"normalizedScore":51.4836,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-sweverified-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.9,"normalizedScore":79.1436,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-livecodebenchv6-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":84.8,"normalizedScore":85.9249,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-swepro-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.1,"normalizedScore":37.9679,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-swemultilingual-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":77.5,"normalizedScore":63.4328,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-nl2repo-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":43.2,"normalizedScore":73.7327,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aascicode-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47,"normalizedScore":77.7403,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-longbenchv2-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":64.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aineedle-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-lcr-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.3,"normalizedScore":86.2616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-critpt-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmmupro-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":70.6,"normalizedScore":24.5161,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mathvision-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-charxiv-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":68.5,"normalizedScore":38.7255,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-videommmu-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.4,"normalizedScore":17.9487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-screenspotpro-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":45.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-vstar-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":67,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aammmupro-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":71.2,"normalizedScore":76.8041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-designarenawebsite-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1277,"normalizedScore":82.0132,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-gpqa-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-supergpqa-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":70.6,"normalizedScore":66.0451,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmlupro-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":89.5,"normalizedScore":99.8577,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmluredux-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":96.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-ceval-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":92.2,"normalizedScore":66.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hle-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":30.8,"normalizedScore":40.5263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aagpqadiamond-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81,"normalizedScore":81.392,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aaomniscienceindex-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-3.9,"normalizedScore":65.3846,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-omniscienceaccuracy-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.7,"normalizedScore":64.433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-omnisciencehallucinationrate-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.4,"normalizedScore":26.0555,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aammlupro-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.9,"normalizedScore":90,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmluprox-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":85.7,"normalizedScore":82.8947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-nova63-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":56.7,"normalizedScore":40,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-ifeval-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":90.9,"normalizedScore":87.8842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-ifbench-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":58,"normalizedScore":42.0601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aaifbench-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43,"normalizedScore":41.1504,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aime2026-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.1,"normalizedScore":93.0248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hmmtfeb2025-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":92.9,"normalizedScore":32.3529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hmmtnov2025-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":93.3,"normalizedScore":53.8462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hmmtfeb2026-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.3,"normalizedScore":83.4595,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmanswerbench-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84,"normalizedScore":42.1488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-frontiermathv2tiers13-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":20.69,"normalizedScore":23.2472,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-frontiermathv2tier4-2026-07-27","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-terminalbench2-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":52.5,"normalizedScore":29.8932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-browsecomp-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":62,"normalizedScore":36.8201,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-claweval-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":56.8,"normalizedScore":71.648,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-qwenclawbench-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":51.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-tau3bench-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":68.4,"normalizedScore":10.8527,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-vitabench-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":43.7,"normalizedScore":87.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-deepplanning-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":37.6,"normalizedScore":48.4342,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-toolathlon-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":36.3,"normalizedScore":19.3018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mcpatlas-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":46.1,"normalizedScore":28.3276,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mcptasks-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-wideresearch-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74,"normalizedScore":67.1498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-tau2bench-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.6,"normalizedScore":96.4682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gertlabs-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":46.76,"normalizedScore":44.6112,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-researchclawbench-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":14.2,"normalizedScore":20.6897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aaagenticindex-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.85,"normalizedScore":35.6065,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-apexagentsaa-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":15.3,"normalizedScore":31.4655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gdpvalaanormalized-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.1,"normalizedScore":33.9706,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gdpvalaa-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":962,"normalizedScore":54.596,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-sweverified-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":76.2,"normalizedScore":72.6519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-livecodebenchv6-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":83.6,"normalizedScore":83.9142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-swepro-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":50.9,"normalizedScore":21.3904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aascicode-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42,"normalizedScore":69.3086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aacodingindex-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.21,"normalizedScore":57.9223,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-longbenchv2-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":63.2,"normalizedScore":93.9086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aineedle-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":68.7,"normalizedScore":50.4673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-lcr-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.7,"normalizedScore":86.79,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-critpt-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.7,"normalizedScore":5.2632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmmupro-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79,"normalizedScore":51.6129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mathvision-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88.6,"normalizedScore":71.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-charxiv-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":80.8,"normalizedScore":68.8725,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-videommmu-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.7,"normalizedScore":25.641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-screenspotpro-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":65.6,"normalizedScore":47.1564,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-vstar-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":95.8,"normalizedScore":96.3211,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aammmupro-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":77.3,"normalizedScore":87.2852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gpqa-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":88.4,"normalizedScore":89.8702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-supergpqa-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":70.4,"normalizedScore":65.7668,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmlupro-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":87.8,"normalizedScore":97.4388,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmluredux-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.9,"normalizedScore":93.5946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-ceval-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":93,"normalizedScore":90.9091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hle-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":28.7,"normalizedScore":36.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aagpqadiamond-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.3,"normalizedScore":93.1818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aaomniscienceindex-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-29.8,"normalizedScore":45.0549,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-omniscienceaccuracy-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.4,"normalizedScore":48.4536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-omnisciencehallucinationrate-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.1,"normalizedScore":9.5296,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmluprox-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":84.7,"normalizedScore":69.7368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-nova63-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-ifeval-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":92.6,"normalizedScore":92.9078,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aaifbench-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":78.8,"normalizedScore":93.9528,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aime2026-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":93.3,"normalizedScore":89.9626,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hmmtfeb2025-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":94.8,"normalizedScore":60.2941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hmmtnov2025-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":92.7,"normalizedScore":46.1538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hmmtfeb2026-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.9,"normalizedScore":87.104,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmanswerbench-2026-07-27","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":80.9,"normalizedScore":16.5289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-terminalbench2-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":60,"normalizedScore":43.2384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-osworldverified-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":72.1,"normalizedScore":71.9565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-mcpatlas-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":57.7,"normalizedScore":48.1229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-toolathlon-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":42.9,"normalizedScore":32.8542,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-tau2bench-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83.3,"normalizedScore":84.0565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aaagenticindex-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.17,"normalizedScore":54.3735,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-apexagentsaa-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":28.2,"normalizedScore":59.2672,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.5,"normalizedScore":49.2647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-gdpvalaa-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1169,"normalizedScore":65.0505,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-vibecodebench-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":47.969,"normalizedScore":67.5591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aacodingindex-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.08,"normalizedScore":69.0459,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aascicode-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49.9,"normalizedScore":82.6307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-frontiercode-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":27,"normalizedScore":9.2466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-lcr-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":91.5456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-critpt-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":10,"normalizedScore":30.9598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-mmmupro-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":76.6,"normalizedScore":43.871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-mmmupropython-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":78,"normalizedScore":56.2914,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aammmupro-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":73.3,"normalizedScore":80.4124,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-gpqa-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":88,"normalizedScore":89.2995,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-hle-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":41.5,"normalizedScore":59.2982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-hlenotools-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":28.2,"normalizedScore":42.7509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aagpqadiamond-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87.5,"normalizedScore":90.625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-18.7,"normalizedScore":53.7677,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.5,"normalizedScore":58.9347,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.8,"normalizedScore":8.6852,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aaifbench-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.3,"normalizedScore":85.8407,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":28.28,"normalizedScore":31.7753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-frontiermathv2tier4-2026-07-27","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.08,"normalizedScore":2.506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-pro-tau2bench-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":87.1,"normalizedScore":87.891,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-gertlabs-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":63.23,"normalizedScore":79.4167,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-jobbench-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":11.4,"normalizedScore":6.2162,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-vibecodebench-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":14.3,"normalizedScore":20.14,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aascicode-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":93.086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aalivecodebench-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aalivecodebench","benchmarkName":"Artificial Analysis LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-arcagi2-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":31.1,"normalizedScore":22.18,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-lcr-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70.7,"normalizedScore":93.395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-critpt-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":9.1,"normalizedScore":28.1734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-mmmupro-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81,"normalizedScore":58.0645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-mathvision-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.6,"normalizedScore":61.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-videommmu-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":87.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-screenspotpro-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":72.7,"normalizedScore":63.981,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-charxiv-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":81.4,"normalizedScore":70.3431,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-vstar-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":88,"normalizedScore":70.2341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aammmupro-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.2,"normalizedScore":92.268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aagpqadiamond-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":90.8,"normalizedScore":95.3125,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aahle-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":37.2,"normalizedScore":67.9283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aaomniscienceindex-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.8,"normalizedScore":80.8477,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-omniscienceaccuracy-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.9,"normalizedScore":90.5498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.9,"normalizedScore":7.3583,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aammlupro-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aaglobalmmlulite-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":92.2,"normalizedScore":90.3846,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aaifbench-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.4,"normalizedScore":81.5634,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-frontiermathv2tiers13-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":37.6,"normalizedScore":42.2472,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-frontiermathv2tier4-2026-07-27","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":18.75,"normalizedScore":22.5904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-terminalbench2-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":56.9,"normalizedScore":37.7224,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-browsecomp-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":73.2,"normalizedScore":60.251,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-hlewithtools-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":45.1,"normalizedScore":28.2051,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-mcpatlas-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":69,"normalizedScore":67.4061,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gdpvalaa-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1189,"normalizedScore":66.0606,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-toolathlon-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":47.8,"normalizedScore":42.9158,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaagenticindex-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.06,"normalizedScore":55.992,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-tau2bench-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95,"normalizedScore":95.8628,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gdpvalaanormalized-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.4,"normalizedScore":50.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aabriefcaseelo-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":833,"normalizedScore":18.1734,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaenterpriseopsgym-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.6,"normalizedScore":49.5614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaharveylab-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.3,"normalizedScore":62.7451,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaitbench-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.5,"normalizedScore":51.1858,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aatau3banking-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.9,"normalizedScore":37.8698,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-terminalbenchhard-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":35.6,"normalizedScore":25.9169,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-codeforces-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-codeforces","benchmarkName":"Codeforces Rating","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":3052,"normalizedScore":60.5128,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-sweverified-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":79,"normalizedScore":76.5193,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-swepro-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.6,"normalizedScore":25.9358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-swemultilingual-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.3,"normalizedScore":52.9851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aacodingindex-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.17,"normalizedScore":69.1731,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aascicode-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":44.9,"normalizedScore":74.199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-mrcr1m-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":78.7,"normalizedScore":91.5641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-corpusqa1m-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":60.5,"normalizedScore":96.7742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-lcr-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63,"normalizedScore":83.2232,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-critpt-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":7.1,"normalizedScore":21.9814,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-designarenawebsite-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1238,"normalizedScore":75.5776,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-mmlupro-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.2,"normalizedScore":95.1622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-simpleqa-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":34.1,"normalizedScore":31.6092,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-chinesesimpleqa-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":78.9,"normalizedScore":57.3643,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gpqa-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":88.1,"normalizedScore":89.4421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gpqadiamond-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":88.1,"normalizedScore":89.4421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-hle-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":34.8,"normalizedScore":47.5439,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaomniscienceindex-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-22.9,"normalizedScore":50.471,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-omniscienceaccuracy-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.2,"normalizedScore":58.4192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-omnisciencehallucinationrate-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":95.8,"normalizedScore":1.4475,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaopennessindex-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50,"normalizedScore":33.4,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaifbench-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":79.2,"normalizedScore":94.5428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-hmmtfeb2026-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.8,"normalizedScore":96.776,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-imoanswerbench-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88.4,"normalizedScore":97.075,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-apex-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":33,"normalizedScore":73.9229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-apexshortlist-2026-07-27","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.7,"normalizedScore":94.4444,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-terminalbench2-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":50.8,"normalizedScore":26.8683,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-browsecomp-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":60.6,"normalizedScore":33.8912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-claweval-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":52.3,"normalizedScore":65.3631,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-qwenclawbench-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":54.3,"normalizedScore":20,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-tau3bench-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":65.7,"normalizedScore":0.3876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-deepsearchqa-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":77.1,"normalizedScore":44.4099,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-deepplanning-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":14.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-toolathlon-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":27.8,"normalizedScore":1.848,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mcpatlas-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":29.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mcptasks-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-wideresearch-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":72.7,"normalizedScore":60.8696,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-tau2bench-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-apexagentsaa-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":11.5,"normalizedScore":23.2759,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gertlabs-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":45.88,"normalizedScore":42.7515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-researchclawbench-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":14,"normalizedScore":18.3908,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-jobbench-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":8.73,"normalizedScore":0.4332,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aaagenticindex-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.69,"normalizedScore":38.9525,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gdpvalaanormalized-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.1,"normalizedScore":36.9118,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gdpvalaa-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1003,"normalizedScore":56.6667,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-sweverified-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":76.8,"normalizedScore":73.4807,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-sweverifiedarcee-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":70.8,"normalizedScore":61.2903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-livecodebenchv6-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":85,"normalizedScore":86.2601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-swepro-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":50.7,"normalizedScore":20.8556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-swemultilingual-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73,"normalizedScore":52.2388,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-swerebench-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.5,"normalizedScore":71.308,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-reactnativeevals-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":77.2,"normalizedScore":24.7012,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-scicode-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":48.7,"normalizedScore":65.5589,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aascicode-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49,"normalizedScore":81.113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aacodingindex-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.78,"normalizedScore":55.9011,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-longbenchv2-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":61,"normalizedScore":82.7411,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-lcr-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.3,"normalizedScore":86.2616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-critpt-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.1,"normalizedScore":9.5975,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmmupro-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-videomme-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-videomme","benchmarkName":"Video-MME","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MME benchmark team","benchmarkVersion":"2024","score":87.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmvu-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":80.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-videommmu-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":86.6,"normalizedScore":74.359,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aammmupro-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.4,"normalizedScore":84.0206,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-designarenawebsite-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1279,"normalizedScore":82.3432,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gpqa-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.6,"normalizedScore":88.7288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gpqadiamond-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87.6,"normalizedScore":88.7288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-supergpqa-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":69.2,"normalizedScore":64.0969,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmlupro-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":87.1,"normalizedScore":96.4428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmluproarcee-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":87.1,"normalizedScore":85.6115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hle-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":30.1,"normalizedScore":39.2982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aaomniscienceindex-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-8.1,"normalizedScore":62.0879,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-omniscienceaccuracy-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.3,"normalizedScore":53.4364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-omnisciencehallucinationrate-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.6,"normalizedScore":39.0832,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmluprox-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":82.3,"normalizedScore":38.1579,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-nova63-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":56,"normalizedScore":22.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-ifeval-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":93.9,"normalizedScore":96.7494,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aaifbench-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.2,"normalizedScore":81.2684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aime2025-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":96.1,"normalizedScore":98.4093,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aime2026-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.8,"normalizedScore":94.2157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aime2025arcee-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":96.3,"normalizedScore":95.3826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hmmtfeb2025-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":95.4,"normalizedScore":69.1176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hmmtnov2025-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":91.1,"normalizedScore":25.641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hmmtfeb2026-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.1,"normalizedScore":85.9826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmanswerbench-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.8,"normalizedScore":23.9669,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-frontiermathv2tiers13-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":27.9,"normalizedScore":31.3483,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-frontiermathv2tier4-2026-07-27","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.2,"normalizedScore":5.0602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-terminalbench2-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":56.4,"normalizedScore":36.8327,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-pinchbench-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-pinchbench","benchmarkName":"PinchBench","benchmarkCategory":"agents","benchmarkOrganisation":"Kilo Code","benchmarkVersion":"2026","score":90,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-browsecomp-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":44.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-tau3bench-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.9,"normalizedScore":20.5426,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gdpvalaanormalized-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.1,"normalizedScore":48.6765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-hlewithtools-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":37.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaagenticindex-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.36,"normalizedScore":49.2635,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-tau2bench-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83.3,"normalizedScore":84.0565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gdpvalaa-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1162,"normalizedScore":64.697,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aabriefcaseelo-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":873,"normalizedScore":21.8635,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaenterpriseopsgym-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.9,"normalizedScore":2.6316,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaharveylab-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.7,"normalizedScore":63.8655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-terminalbenchhard-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":36.4,"normalizedScore":27.8729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-sweverified-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":71.9,"normalizedScore":66.7127,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-swemultilingual-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":67.7,"normalizedScore":39.0547,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-scicode-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":44.6,"normalizedScore":53.1722,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aacodingindex-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.27,"normalizedScore":59.4205,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aascicode-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.9,"normalizedScore":65.7673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-lcr-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67,"normalizedScore":88.5073,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-critpt-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.1,"normalizedScore":9.5975,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-longbenchv2-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":61.9,"normalizedScore":87.3096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-designarenawebsite-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1129,"normalizedScore":57.5908,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gpqa-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gpqadiamond-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-hle-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":26.7,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-hlenotools-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":26.7,"normalizedScore":39.9628,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-mmlupro-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.8,"normalizedScore":96.0159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-omniscienceaccuracy-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.6,"normalizedScore":31.6151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaomniscienceindex-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-0.8,"normalizedScore":67.8179,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-omnisciencehallucinationrate-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.5,"normalizedScore":82.6297,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaopennessindex-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.3,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-mmluprox-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":83,"normalizedScore":47.3684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-ifbench-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":81.7,"normalizedScore":92.9185,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaifbench-2026-07-27","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":81.4,"normalizedScore":97.7876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-terminalbench2-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":49.4,"normalizedScore":24.3772,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-browsecomp-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":63.8,"normalizedScore":40.5858,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-osworldverified-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":58,"normalizedScore":41.3043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-tau2bench-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.6,"normalizedScore":94.4501,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aaagenticindex-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.72,"normalizedScore":37.1886,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-gdpvalaanormalized-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.1,"normalizedScore":35.4412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-gdpvalaa-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":982,"normalizedScore":55.6061,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-sweverified-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":72,"normalizedScore":66.8508,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aacodingindex-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.71,"normalizedScore":54.3887,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aascicode-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42,"normalizedScore":69.3086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-longbenchv2-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.2,"normalizedScore":78.6802,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-lcr-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-critpt-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmmu-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":83.9,"normalizedScore":96.0623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmvu-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":74.7,"normalizedScore":29.6296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mathvision-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":59.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-charxiv-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":77.2,"normalizedScore":60.049,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-vstar-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":93.2,"normalizedScore":87.6254,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aammmupro-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75,"normalizedScore":83.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmlupro-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.7,"normalizedScore":95.8736,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-supergpqa-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":67.1,"normalizedScore":61.1745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-gpqa-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":86.6,"normalizedScore":87.302,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aagpqadiamond-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.7,"normalizedScore":88.0682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aahle-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.4,"normalizedScore":40.4382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aaomniscienceindex-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-39.6,"normalizedScore":37.3626,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-omniscienceaccuracy-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.7,"normalizedScore":36.9416,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-omnisciencehallucinationrate-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.5,"normalizedScore":13.8721,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmluprox-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":82.2,"normalizedScore":36.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-ifeval-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":93.4,"normalizedScore":95.2719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aaifbench-2026-07-27","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.7,"normalizedScore":89.3805,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-terminalbench2-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":41.6,"normalizedScore":10.4982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-browsecomp-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":61,"normalizedScore":34.728,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-osworldverified-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":56.2,"normalizedScore":37.3913,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-tau2bench-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.9,"normalizedScore":94.7528,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-gertlabs-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.41,"normalizedScore":29.0786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-sweverified-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":72.4,"normalizedScore":67.4033,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-swerebench-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.9,"normalizedScore":72.9958,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aascicode-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.5,"normalizedScore":65.0927,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-longbenchv2-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.6,"normalizedScore":80.7107,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-lcr-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.3,"normalizedScore":88.9036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-critpt-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmmu-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":82.3,"normalizedScore":93.0621,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmvu-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":73.3,"normalizedScore":12.3457,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mathvision-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86,"normalizedScore":58.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-vstar-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":93.7,"normalizedScore":89.2977,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aammmupro-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75,"normalizedScore":83.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmlupro-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.1,"normalizedScore":95.0199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-supergpqa-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":65.6,"normalizedScore":59.0871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-gpqa-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":85.5,"normalizedScore":85.7326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aagpqadiamond-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.8,"normalizedScore":88.2102,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aahle-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":22.2,"normalizedScore":38.0478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aaomniscienceindex-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-42,"normalizedScore":35.4788,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-omniscienceaccuracy-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21,"normalizedScore":30.5842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-omnisciencehallucinationrate-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":79.7,"normalizedScore":20.8685,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmluprox-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":82.2,"normalizedScore":36.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-ifeval-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aaifbench-2026-07-27","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.6,"normalizedScore":89.233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-browsecomp-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":65.8,"normalizedScore":44.7699,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-osworldverified-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":47.3,"normalizedScore":18.0435,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-tau2bench-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-gertlabs-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":46.54,"normalizedScore":44.1462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-jobbench-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":34.3,"normalizedScore":55.8155,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-sweverified-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80,"normalizedScore":77.9006,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-swepro-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.6,"normalizedScore":33.9572,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-vibecodebench-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":53.499,"normalizedScore":75.3475,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aascicode-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":52.1,"normalizedScore":86.3406,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-arcagi2-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":52.9,"normalizedScore":49.8099,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-lcr-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.7,"normalizedScore":96.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-critpt-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":11.6,"normalizedScore":35.9133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-mmmupro-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79.5,"normalizedScore":53.2258,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-mathvision-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83,"normalizedScore":43.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-charxiv-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":82.1,"normalizedScore":72.0588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-vstar-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":75.9,"normalizedScore":29.7659,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-designarenawebsite-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1224,"normalizedScore":73.2673,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-gpqa-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.4,"normalizedScore":95.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aagpqadiamond-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":90.3,"normalizedScore":94.6023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aahle-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":35.4,"normalizedScore":64.3426,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-1,"normalizedScore":67.6609,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.8,"normalizedScore":69.7595,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":79.7,"normalizedScore":20.8685,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aaifbench-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.4,"normalizedScore":88.9381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aaaime2025-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaaime2025","benchmarkName":"Artificial Analysis AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":40.7,"normalizedScore":45.7303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-frontiermathv2tier4-2026-07-27","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":18.8,"normalizedScore":22.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-terminalbench2-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59.3,"normalizedScore":41.9929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-claweval-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":72.4,"normalizedScore":93.4358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-qwenclawbench-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":53.4,"normalizedScore":12.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-qwenwebbench-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1487,"normalizedScore":52.6316,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-androidworld-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":70.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aaagenticindex-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.03,"normalizedScore":48.6634,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-tau2bench-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.2,"normalizedScore":95.0555,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gdpvalaanormalized-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.9,"normalizedScore":46.9118,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gdpvalaa-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1138,"normalizedScore":63.4848,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gertlabs-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":54.84,"normalizedScore":61.6864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-sweverified-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.2,"normalizedScore":74.0331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-swemultilingual-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":71.3,"normalizedScore":48.01,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-swepro-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":53.5,"normalizedScore":28.3422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-livecodebench-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":83.9,"normalizedScore":85.7407,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-nl2repo-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":36.2,"normalizedScore":41.4747,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aacodingindex-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":53.72,"normalizedScore":65.7102,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aascicode-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.8,"normalizedScore":65.5987,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-lcr-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.7,"normalizedScore":90.753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-critpt-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmmu-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":82.9,"normalizedScore":94.1871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmmupro-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":75.8,"normalizedScore":41.2903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-realworldqa-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84.1,"normalizedScore":90.1651,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-dynamath-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-dynamath","benchmarkName":"DynaMath","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mstar-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mstar","benchmarkName":"MStar","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-simplevqa-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":56.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-charxiv-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":78.4,"normalizedScore":62.9902,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-ccocr-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ccocr","benchmarkName":"CC-OCR","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-countbench-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-countbench","benchmarkName":"CountBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":97.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-refcocoavg-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":92.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-erqa-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":62.5,"normalizedScore":59.8901,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-videommewithsub-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.7,"normalizedScore":88.4615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-videommmu-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.4,"normalizedScore":17.9487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mlvuavg-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mlvuavg","benchmarkName":"MLVU mean average","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.6,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-vstar-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":94.7,"normalizedScore":92.6421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aammmupro-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.6,"normalizedScore":82.646,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmlupro-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.2,"normalizedScore":95.1622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmluredux-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":93.5,"normalizedScore":88.3195,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-supergpqa-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":66,"normalizedScore":59.6438,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-ceval-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":91.4,"normalizedScore":42.4242,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gpqa-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.8,"normalizedScore":89.0141,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hle-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":24,"normalizedScore":28.5965,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aagpqadiamond-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.2,"normalizedScore":85.9375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aaomniscienceindex-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-19.8,"normalizedScore":52.9042,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-omniscienceaccuracy-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.2,"normalizedScore":27.4914,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-omnisciencehallucinationrate-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.3,"normalizedScore":58.7455,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aaifbench-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":67.6,"normalizedScore":77.4336,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hmmtfeb2025-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":93.8,"normalizedScore":45.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hmmtnov2025-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":90.7,"normalizedScore":20.5128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hmmtfeb2026-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84.3,"normalizedScore":82.0578,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmanswerbench-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":80.8,"normalizedScore":15.7025,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aime2026-2026-07-27","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":94.1,"normalizedScore":91.3236,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-terminalbench2-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":46,"normalizedScore":18.3274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-sweverified-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.5,"normalizedScore":68.9227,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-swepro-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.8,"normalizedScore":26.4706,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-graphwalksbfs128k-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-graphwalksbfs128k","benchmarkName":"Graphwalks BFS 0K-128K","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-gpqa-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":84.2,"normalizedScore":83.8779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-gpqadiamond-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":84.2,"normalizedScore":83.8779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-mmlupro-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85,"normalizedScore":93.4548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-simpleqa-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":31,"normalizedScore":22.7011,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-ifbench-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":85,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-aime2025-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":97,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-aime2026-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":94.5,"normalizedScore":92.0041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-hmmtfeb2026-2026-07-27","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84.9,"normalizedScore":82.8988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-terminalbench2-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59.1,"normalizedScore":41.637,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-mcpatlas-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":69.4,"normalizedScore":68.0887,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-toolathlon-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":46.3,"normalizedScore":39.8357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-claweval-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":59.8,"normalizedScore":75.838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-gertlabs-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":50.28,"normalizedScore":52.0499,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-researchclawbench-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":17.1,"normalizedScore":54.023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-sweverified-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.6,"normalizedScore":69.0608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-swepro-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.1,"normalizedScore":24.5989,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-swemultilingual-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":69.8,"normalizedScore":44.2786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-mrcr1m-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":44.7,"normalizedScore":31.8102,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-corpusqa1m-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":35.6,"normalizedScore":43.2258,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-designarenawebsite-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1264,"normalizedScore":79.868,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-mmlupro-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":82.9,"normalizedScore":90.4667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-simpleqa-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":45,"normalizedScore":62.931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-chinesesimpleqa-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":75.8,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-gpqa-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":72.9,"normalizedScore":67.7557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-gpqadiamond-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":72.9,"normalizedScore":67.7557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-hle-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":7.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-hmmtfeb2026-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":31.7,"normalizedScore":8.3263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-imoanswerbench-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":35.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-apex-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":0.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-apexshortlist-2026-07-27","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":9.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-terminalbench2-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":40.5,"normalizedScore":8.5409,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-browsecomp-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":61,"normalizedScore":34.728,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-osworldverified-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":54.5,"normalizedScore":33.6957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-tau2bench-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":89.2,"normalizedScore":90.0101,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-gertlabs-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":28.96,"normalizedScore":6.9949,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-sweverified-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":69.2,"normalizedScore":62.9834,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-swerebench-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":53.7,"normalizedScore":51.0549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aascicode-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.7,"normalizedScore":62.0573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-longbenchv2-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":59,"normalizedScore":72.5888,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-lcr-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.7,"normalizedScore":82.8269,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-critpt-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmmu-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":81.4,"normalizedScore":91.3745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmvu-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":72.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mathvision-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.9,"normalizedScore":48,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-vstar-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":92.7,"normalizedScore":85.9532,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aammmupro-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.7,"normalizedScore":79.3814,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmlupro-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.3,"normalizedScore":93.8816,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-supergpqa-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":63.4,"normalizedScore":56.0256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-gpqa-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":84.2,"normalizedScore":83.8779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aagpqadiamond-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.5,"normalizedScore":86.3636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aahle-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":19.7,"normalizedScore":33.0677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aaomniscienceindex-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-46.4,"normalizedScore":32.0251,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-omniscienceaccuracy-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.5,"normalizedScore":29.7251,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-omnisciencehallucinationrate-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84,"normalizedScore":15.6815,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmluprox-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":81,"normalizedScore":21.0526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-ifeval-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":91.9,"normalizedScore":90.8392,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aaifbench-2026-07-27","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.5,"normalizedScore":84.6608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-terminalbench2-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":51.5,"normalizedScore":28.1139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-claweval-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":68.7,"normalizedScore":88.2682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-qwenclawbench-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":52.6,"normalizedScore":6.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-qwenwebbench-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1397,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-tau3bench-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":67.2,"normalizedScore":6.2016,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-vitabench-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":35.6,"normalizedScore":62.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-deepplanning-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":25.9,"normalizedScore":24.0084,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-toolathlon-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":26.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mcpatlas-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":62.8,"normalizedScore":56.8259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-wideresearch-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":60.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aaagenticindex-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.41,"normalizedScore":38.4434,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-tau2bench-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.3,"normalizedScore":96.1655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gdpvalaanormalized-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.6,"normalizedScore":40.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gdpvalaa-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1052,"normalizedScore":59.1414,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gertlabs-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":42.65,"normalizedScore":35.9256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-sweverified-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.4,"normalizedScore":68.7845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-swemultilingual-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":67.2,"normalizedScore":37.8109,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-swepro-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":49.5,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-livecodebench-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":80.4,"normalizedScore":79.2593,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-nl2repo-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":29.4,"normalizedScore":10.1382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aacodingindex-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.88,"normalizedScore":48.9753,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aascicode-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.8,"normalizedScore":58.8533,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-lcr-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.7,"normalizedScore":84.148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-critpt-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmmu-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":81.7,"normalizedScore":91.937,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmmupro-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":75.3,"normalizedScore":39.6774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-realworldqa-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.3,"normalizedScore":94.38,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-omnidocbench15-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnidocbench15","benchmarkName":"OmniDocBench 1.5","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":89.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-charxiv-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":78,"normalizedScore":62.0098,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-simplevqa-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":58.9,"normalizedScore":10.9375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-ccocr-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ccocr","benchmarkName":"CC-OCR","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-ai2dtest-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ai2dtest","benchmarkName":"AI2D test split","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-refcocoavg-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":92,"normalizedScore":95.1923,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-odinw13-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-odinw13","benchmarkName":"ODINW13","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":50.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-videommewithsub-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.6,"normalizedScore":46.1538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-videommenosub-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommenosub","benchmarkName":"Video-MME without subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":82.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-videommmu-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":83.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mlvuavg-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mlvuavg","benchmarkName":"MLVU mean average","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aammmupro-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75,"normalizedScore":83.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmlupro-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.2,"normalizedScore":93.7393,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-supergpqa-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":64.7,"normalizedScore":57.8347,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-ceval-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":90,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gpqa-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":86,"normalizedScore":86.446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hle-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":21.4,"normalizedScore":24.0351,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aagpqadiamond-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.1,"normalizedScore":85.7955,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aaomniscienceindex-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-21.4,"normalizedScore":51.6484,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-omniscienceaccuracy-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.9,"normalizedScore":26.9759,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-omnisciencehallucinationrate-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.7,"normalizedScore":57.0567,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aaifbench-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":64.4,"normalizedScore":72.7139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2025-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":90.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hmmtnov2025-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":89.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2026-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.6,"normalizedScore":81.0765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmanswerbench-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":78.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aime2026-2026-07-27","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":92.7,"normalizedScore":88.9418,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-terminalbench2-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":41,"normalizedScore":9.4306,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-browsecomp-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":52,"normalizedScore":15.8996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-vitabench-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":15.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aaagenticindex-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.39,"normalizedScore":45.681,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-tau2bench-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gertlabs-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.95,"normalizedScore":30.2198,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gdpvalaanormalized-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":48.9706,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gdpvalaa-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1166,"normalizedScore":64.899,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-sweverified-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.8,"normalizedScore":69.337,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-livecodebench-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":84.9,"normalizedScore":87.5926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-swerebench-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.7,"normalizedScore":72.1519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aacodingindex-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.26,"normalizedScore":53.7527,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aascicode-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.1,"normalizedScore":74.5363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aalivecodebench-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aalivecodebench","benchmarkName":"Artificial Analysis LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.4,"normalizedScore":41.0256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-lcr-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64,"normalizedScore":84.5443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-critpt-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.7,"normalizedScore":5.2632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-designarenawebsite-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1255,"normalizedScore":78.3828,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gpqa-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":85.7,"normalizedScore":86.018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-mmlupro-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":84.3,"normalizedScore":92.4587,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-hle-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":24.8,"normalizedScore":30,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aagpqadiamond-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.9,"normalizedScore":88.3523,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aaomniscienceindex-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-34.6,"normalizedScore":41.2873,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-omniscienceaccuracy-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.3,"normalizedScore":44.8454,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-omnisciencehallucinationrate-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.3,"normalizedScore":8.082,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aaifbench-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":67.9,"normalizedScore":77.8761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aime2025-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":95.7,"normalizedScore":97.7024,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-frontiermathv2tiers13-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.439,"normalizedScore":2.7404,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-frontiermathv2tier4-2026-07-27","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-terminalbench2-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":46.3,"normalizedScore":18.8612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-osworldverified-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":39,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-mcpatlas-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":56.1,"normalizedScore":45.3925,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-toolathlon-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":35.5,"normalizedScore":17.6591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-tau2bench-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":76,"normalizedScore":76.6902,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aaagenticindex-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.54,"normalizedScore":49.5908,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-apexagentsaa-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":24.9,"normalizedScore":52.1552,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.1,"normalizedScore":44.2647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-gdpvalaa-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1101,"normalizedScore":61.6162,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-vibecodebench-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":26.097,"normalizedScore":36.7548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aacodingindex-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.07,"normalizedScore":69.0318,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aascicode-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.9,"normalizedScore":77.5717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-lcr-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66,"normalizedScore":87.1863,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-critpt-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":9.3,"normalizedScore":28.7926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-mmmupro-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":66.1,"normalizedScore":10,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-mmmupropython-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":69.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aammmupro-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":65.4,"normalizedScore":66.8385,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-gpqa-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":82.8,"normalizedScore":81.8804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-hle-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":37.7,"normalizedScore":52.6316,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-hlenotools-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":24.3,"normalizedScore":35.5019,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aagpqadiamond-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81.7,"normalizedScore":82.3864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-29.5,"normalizedScore":45.2904,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.4,"normalizedScore":38.1443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.6,"normalizedScore":28.2268,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aaifbench-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.9,"normalizedScore":89.6755,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":25.86,"normalizedScore":29.0562,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-frontiermathv2tier4-2026-07-27","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":6.25,"normalizedScore":7.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-terminalbench2-2026-07-27","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":82.1,"normalizedScore":82.5623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-swepro-2026-07-27","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":73.7,"normalizedScore":82.3529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-livecodebenchv6-2026-07-27","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":93.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-livecodebenchpro-2026-07-27","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":90.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-scicode-2026-07-27","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":58.7,"normalizedScore":95.7704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-mrcrv2-2026-07-27","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":93.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-charxiv-2026-07-27","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":86.6,"normalizedScore":83.0882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-gpqa-2026-07-27","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-gpqadiamond-2026-07-27","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-hlenotools-2026-07-27","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":50,"normalizedScore":83.2714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-osworldverified-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":83,"normalizedScore":95.6522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-gdpvalaa-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1423,"normalizedScore":77.8788,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aaagenticindex-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.72,"normalizedScore":69.9218,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-gdpvalaanormalized-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.2,"normalizedScore":67.9412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aabriefcaseelo-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":964,"normalizedScore":30.2583,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aatau3banking-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.5,"normalizedScore":47.3373,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aaterminalbench21-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.5,"normalizedScore":53.7849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aaautomationbench-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.1,"normalizedScore":92,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-deepswe-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":49,"normalizedScore":26.6254,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aacodingindex-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.24,"normalizedScore":87.6466,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aascicode-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":52.7,"normalizedScore":87.3524,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-lcr-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-critpt-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":10.6,"normalizedScore":32.8173,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aammmupro-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":83.2,"normalizedScore":97.4227,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aagpqadiamond-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":92.8,"normalizedScore":98.1534,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aahle-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":38.3,"normalizedScore":70.1195,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aaomniscienceindex-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.5,"normalizedScore":86.8917,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-omniscienceaccuracy-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.2,"normalizedScore":80.756,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":53.5,"normalizedScore":52.4729,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-terminalbench2-2026-07-27","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":80.2,"normalizedScore":79.1815,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-swepro-2026-07-27","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":59,"normalizedScore":43.0481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-livecodebenchv6-2026-07-27","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":92.9,"normalizedScore":99.4973,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-livecodebenchpro-2026-07-27","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":87.8,"normalizedScore":95.596,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-scicode-2026-07-27","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":60.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-mrcrv2-2026-07-27","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":86.6,"normalizedScore":86.0558,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-charxiv-2026-07-27","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":85.1,"normalizedScore":79.4118,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-gpqa-2026-07-27","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-gpqadiamond-2026-07-27","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-hlenotools-2026-07-27","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":47.2,"normalizedScore":78.0669,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-swe-1-7-terminalbench2-2026-07-27","modelSlug":"swe-1-7","modelName":"SWE-1.7","providerId":"cognition","providerName":"Cognition","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":81.5,"normalizedScore":81.4947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-swe-1-7-frontiercode-2026-07-27","modelSlug":"swe-1-7","modelName":"SWE-1.7","providerId":"cognition","providerName":"Cognition","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":42.3,"normalizedScore":61.6438,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-swe-1-7-swemultilingual-2026-07-27","modelSlug":"swe-1-7","modelName":"SWE-1.7","providerId":"cognition","providerName":"Cognition","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":77.8,"normalizedScore":64.1791,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-35b-a3b-osworldverified-2026-07-27","modelSlug":"holo3-35b-a3b","modelName":"Holo3-35B-A3B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":82.56,"normalizedScore":94.6957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-122b-a10b-osworldverified-2026-07-27","modelSlug":"holo3-122b-a10b","modelName":"Holo3-122B-A10B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":78.85,"normalizedScore":86.6304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-tau2bench-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":4.1,"normalizedScore":4.1372,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aascicode-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":25.2,"normalizedScore":40.9781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-lcr-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8,"normalizedScore":10.568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-critpt-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-mmlupro-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":81.8,"normalizedScore":88.9015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aagpqadiamond-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":62.8,"normalizedScore":55.5398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aahle-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.9,"normalizedScore":3.5857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aaomniscienceindex-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-62.3,"normalizedScore":19.5447,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-omniscienceaccuracy-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.4,"normalizedScore":12.3711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-omnisciencehallucinationrate-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81,"normalizedScore":19.3004,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aaifbench-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":33.5,"normalizedScore":27.1386,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aime2025-2026-07-27","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":85.3,"normalizedScore":79.3213,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-terminalbench2-2026-07-27","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":77.5,"normalizedScore":74.3772,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-claweval-2026-07-27","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":77.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-sweverified-2026-07-27","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":82.4,"normalizedScore":81.2155,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-swepro-2026-07-27","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":62.2,"normalizedScore":51.6043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-swemultilingual-2026-07-27","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.9,"normalizedScore":66.9154,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-nl2repo-2026-07-27","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":48.2,"normalizedScore":96.7742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-terminalbench2-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":83.3,"normalizedScore":84.6975,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-deepswe-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":53,"normalizedScore":39.0093,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaagenticindex-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.69,"normalizedScore":82.5968,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-gdpvalaanormalized-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.4,"normalizedScore":75.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-gdpvalaa-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1527,"normalizedScore":83.1313,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aabriefcaseelo-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1318,"normalizedScore":62.9151,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaautomationbench-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.4,"normalizedScore":93.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaenterpriseopsgym-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.8,"normalizedScore":54.8246,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaharveylab-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":92.4,"normalizedScore":93.8375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aatau3banking-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.6,"normalizedScore":95.2663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaterminalbench21-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.6,"normalizedScore":70.1195,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-swepro-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":64.7,"normalizedScore":58.2888,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-swemultilingual-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78,"normalizedScore":64.6766,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-vulcanbench-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vulcanbench","benchmarkName":"VulcanBench v3","benchmarkCategory":"coding","benchmarkOrganisation":"VulcanBench contributors","benchmarkVersion":"2026","score":91.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aacodingindex-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.45,"normalizedScore":92.1837,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aascicode-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":54.1,"normalizedScore":89.7133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-arcagi2-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":52.64,"normalizedScore":49.4804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-arcagi3-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.6984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-lcr-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.7,"normalizedScore":89.432,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-critpt-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":15.4,"normalizedScore":47.678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aammmupro-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.4,"normalizedScore":92.6117,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-designarenawebsite-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1325,"normalizedScore":89.934,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aagpqadiamond-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":93.1,"normalizedScore":98.5795,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aahle-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":40.3,"normalizedScore":74.1036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaomniscienceindex-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.4,"normalizedScore":89.168,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-omniscienceaccuracy-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.1,"normalizedScore":84.0206,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-omnisciencehallucinationrate-2026-07-27","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":53.5,"normalizedScore":52.4729,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-browsecomp-2026-07-27","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":75.51,"normalizedScore":65.0837,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-hlewithtools-2026-07-27","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":47.6,"normalizedScore":37.3626,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-vitabench-2026-07-27","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":38.75,"normalizedScore":71.7593,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-longbenchv2-2026-07-27","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.2,"normalizedScore":78.6802,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-hle-2026-07-27","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":47.6,"normalizedScore":70,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-ifeval-2026-07-27","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.82,"normalizedScore":99.4681,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-celeris-1-mmlupro-2026-07-27","modelSlug":"celeris-1","modelName":"Celeris-1","providerId":"celeris","providerName":"Celeris","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":75.9,"normalizedScore":80.5065,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Celeris-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-tau2bench-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74,"normalizedScore":74.672,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-gertlabs-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":65.59,"normalizedScore":84.4041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-researchclawbench-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":20.7,"normalizedScore":95.4023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-osworld2-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":13.9,"normalizedScore":16.3717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-vibecodebench-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":71.003,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-reactnativeevals-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":82.8,"normalizedScore":47.012,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aascicode-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50.1,"normalizedScore":82.968,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-frontiercode-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":38.5,"normalizedScore":48.6301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-lcr-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67,"normalizedScore":88.5073,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-critpt-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.1,"normalizedScore":15.7895,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aammmupro-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":76.4,"normalizedScore":85.7388,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-designarenawebsite-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1325,"normalizedScore":89.934,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aagpqadiamond-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":88.5,"normalizedScore":92.0455,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aahle-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":31.2,"normalizedScore":55.9761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aaomniscienceindex-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.2,"normalizedScore":79.5918,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-omniscienceaccuracy-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.5,"normalizedScore":69.244,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-omnisciencehallucinationrate-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.9,"normalizedScore":54.4029,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aaifbench-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43.6,"normalizedScore":42.0354,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-frontiermathv2tiers13-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":43.793,"normalizedScore":49.2056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-frontiermathv2tier4-2026-07-27","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":22.917,"normalizedScore":27.6108,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-claweval-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":40.2,"normalizedScore":48.4637,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-vitabench-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":18.5,"normalizedScore":9.2593,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-tau2bench-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":78.9,"normalizedScore":79.6165,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-gertlabs-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":29.57,"normalizedScore":8.284,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-swerebench-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":60.9,"normalizedScore":81.4346,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-reactnativeevals-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71.5,"normalizedScore":1.992,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aascicode-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.7,"normalizedScore":63.7437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-lcr-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39,"normalizedScore":51.5192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-critpt-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-designarenawebsite-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1204,"normalizedScore":69.967,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aagpqadiamond-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":75.1,"normalizedScore":73.0114,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aahle-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":10.5,"normalizedScore":14.741,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aaomniscienceindex-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-46.7,"normalizedScore":31.7896,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-omniscienceaccuracy-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.2,"normalizedScore":36.0825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-omnisciencehallucinationrate-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.5,"normalizedScore":4.222,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aaifbench-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":49,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-frontiermathv2tiers13-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":22.1,"normalizedScore":24.8315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-frontiermathv2tier4-2026-07-27","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.1,"normalizedScore":2.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-5-terminalbench2-2026-07-27","modelSlug":"composer-2-5","modelName":"Composer 2.5","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":69.3,"normalizedScore":59.7865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-5-swemultilingual-2026-07-27","modelSlug":"composer-2-5","modelName":"Composer 2.5","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":79.8,"normalizedScore":69.1542,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-terminalbench2-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":77.3,"normalizedScore":74.0214,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-osworldverified-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":64.7,"normalizedScore":55.8696,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-tau2bench-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86,"normalizedScore":86.781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-gertlabs-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":57.47,"normalizedScore":67.2443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-jobbench-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":33.7,"normalizedScore":54.5159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-sweverified-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":85,"normalizedScore":84.8066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-swepro-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.8,"normalizedScore":37.1658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-swerebench-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.2,"normalizedScore":70.0422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-vibecodebench-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":61.767,"normalizedScore":86.9921,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aascicode-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.2,"normalizedScore":88.1956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-lcr-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-critpt-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":16.9,"normalizedScore":52.322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aammmupro-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.5,"normalizedScore":89.3471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-designarenawebsite-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1193,"normalizedScore":68.1518,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aagpqadiamond-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":91.5,"normalizedScore":96.3068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aahle-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":39.9,"normalizedScore":73.3068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.9,"normalizedScore":76.2166,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.8,"normalizedScore":83.5052,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.9,"normalizedScore":12.1834,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aaifbench-2026-07-27","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.4,"normalizedScore":88.9381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-tau2bench-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":62.6,"normalizedScore":63.1685,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aacodingindex-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.72,"normalizedScore":45.9223,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aascicode-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.8,"normalizedScore":58.8533,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-lcr-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59.3,"normalizedScore":78.3355,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-critpt-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-mmlu-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":91.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-gpqa-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":75.7,"normalizedScore":71.7506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aagpqadiamond-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":74.7,"normalizedScore":72.4432,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aahle-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7.7,"normalizedScore":9.1633,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aaomniscienceindex-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.5,"normalizedScore":60.2041,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-omniscienceaccuracy-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.7,"normalizedScore":54.1237,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-omnisciencehallucinationrate-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":33.4138,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-ifeval-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":92.2,"normalizedScore":91.7258,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aaifbench-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.3,"normalizedScore":81.4159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-frontiermathv2tiers13-2026-07-27","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":9.31,"normalizedScore":10.4607,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-bfclv4-2026-07-27","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":39.22,"normalizedScore":33.7039,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-livecodebenchv6-2026-07-27","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":65.8,"normalizedScore":54.0885,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-gpqa-2026-07-27","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":71,"normalizedScore":65.0449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-gpqadiamond-2026-07-27","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":71,"normalizedScore":65.0449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-mmlupro-2026-07-27","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":74.2,"normalizedScore":78.0876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-ifeval-2026-07-27","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":85.58,"normalizedScore":72.1631,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-ifbench-2026-07-27","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":52.56,"normalizedScore":30.3863,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-aime2026-2026-07-27","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":89.1,"normalizedScore":82.8173,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-hmmtfeb2026-2026-07-27","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":71.6,"normalizedScore":64.2557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-imoanswerbench-2026-07-27","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":59.3,"normalizedScore":43.8757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-apex-2026-07-27","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":32.2,"normalizedScore":72.1088,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-terminalbench2-2026-07-27","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":61.7,"normalizedScore":46.2633,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-swemultilingual-2026-07-27","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.7,"normalizedScore":53.9801,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-swerebench-2026-07-27","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58,"normalizedScore":69.1983,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-reactnativeevals-2026-07-27","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":96.1,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-tau3bench-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":91.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaagenticindex-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19,"normalizedScore":34.0607,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-tau2bench-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.2,"normalizedScore":95.0555,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-gdpvalaanormalized-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.6,"normalizedScore":31.7647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-gdpvalaa-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":933,"normalizedScore":53.1313,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-gertlabs-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.1,"normalizedScore":28.4235,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaenterpriseopsgym-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.7,"normalizedScore":23.6842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaharveylab-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.1,"normalizedScore":28.5714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-terminalbenchhard-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":20.2934,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-sweverified-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.6,"normalizedScore":74.5856,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aacodingindex-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.9,"normalizedScore":56.0707,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aascicode-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.6,"normalizedScore":65.2614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-lcr-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61,"normalizedScore":80.5812,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-critpt-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aammmupro-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":64.9,"normalizedScore":65.9794,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aagpqadiamond-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":74.8,"normalizedScore":72.5852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aahle-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":12.8,"normalizedScore":19.3227,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaomniscienceindex-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-36.3,"normalizedScore":39.9529,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-omniscienceaccuracy-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.1,"normalizedScore":37.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-omnisciencehallucinationrate-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82,"normalizedScore":18.0941,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaopennessindex-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaifbench-2026-07-27","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":68.8,"normalizedScore":79.2035,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-spider2lite-2026-07-27","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-spider2lite","benchmarkName":"Spider 2.0-Lite","benchmarkCategory":"coding","benchmarkOrganisation":"Spider 2.0 authors","benchmarkVersion":"2024","score":52.9,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-ocrbenchv2-2026-07-27","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"ocrbench-v2","benchmarkName":"OCRBench v2","benchmarkCategory":"multimodal","benchmarkOrganisation":"OCRBench authors","benchmarkVersion":"2025","score":70.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-olmocr-2026-07-27","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-olmocr","benchmarkName":"olmOCR-Bench","benchmarkCategory":"multimodal","benchmarkOrganisation":"Allen Institute for AI","benchmarkVersion":"2025","score":85.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-refcocoavg-2026-07-27","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":82.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-voxpopuliwer-2026-07-27","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-voxpopuliwer","benchmarkName":"VoxPopuli-Cleaned-AA Word Error Rate","benchmarkCategory":"multimodal","benchmarkOrganisation":"Artificial Analysis / VoxPopuli dataset authors","benchmarkVersion":"2026","score":2.4,"normalizedScore":50,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-mmmupro-2026-07-27","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":71.1,"normalizedScore":26.129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-gpqa-2026-07-27","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":89.9,"normalizedScore":92.0103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-gpqadiamond-2026-07-27","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":89.9,"normalizedScore":92.0103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-mmmlu-2026-07-27","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-sobvalueacc-2026-07-27","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-sobvalueacc","benchmarkName":"Structured Output Benchmark Value Accuracy","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Interfaze","benchmarkVersion":"2026","score":79.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-terminalbench2-2026-07-27","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":70.2,"normalizedScore":61.3879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-toolathlonverified-2026-07-27","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-toolathlonverified","benchmarkName":"Toolathlon-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":49.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-swemultilingual-2026-07-27","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.5,"normalizedScore":65.9204,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-swepro-2026-07-27","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":59.4,"normalizedScore":44.1176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-deepswe-2026-07-27","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":40.4,"normalizedScore":0,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-terminalbench2-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":65.4,"normalizedScore":52.847,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-qwenclawbench-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59,"normalizedScore":57.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-qwenwebbench-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1532,"normalizedScore":78.9474,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-tau2bench-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-swepro-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.3,"normalizedScore":38.5027,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-scicode-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":47,"normalizedScore":60.423,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-nl2repo-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":42.9,"normalizedScore":72.3502,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aascicode-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.9,"normalizedScore":77.5717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-lcr-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-critpt-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.7,"normalizedScore":11.4551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-supergpqa-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":73.9,"normalizedScore":70.6374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aagpqadiamond-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":88.8,"normalizedScore":92.4716,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aahle-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.9,"normalizedScore":51.3944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aaomniscienceindex-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.2,"normalizedScore":76.4521,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-omniscienceaccuracy-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.7,"normalizedScore":59.2784,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-omnisciencehallucinationrate-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.2,"normalizedScore":63.6912,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aaifbench-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.6,"normalizedScore":90.708,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-frontiermathv2tiers13-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":23.103,"normalizedScore":25.9584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-frontiermathv2tier4-2026-07-27","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-claweval-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":63.8,"normalizedScore":81.4246,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-gdpvalaa-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1265,"normalizedScore":69.899,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-tau3bench-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":72.9,"normalizedScore":28.2946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-terminalbench2-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":68.4,"normalizedScore":58.1851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaagenticindex-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.11,"normalizedScore":52.4459,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-tau2bench-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.2,"normalizedScore":95.0555,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-gdpvalaanormalized-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.3,"normalizedScore":56.3235,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-apexagentsaa-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":2.4,"normalizedScore":3.6638,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-gertlabs-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":62.7,"normalizedScore":78.2967,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aabriefcaseelo-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":878,"normalizedScore":22.3247,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaitbench-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.2,"normalizedScore":64.4269,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-terminalbenchhard-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":43.2,"normalizedScore":44.4988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaterminalbench21-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.2,"normalizedScore":4.7809,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaharveylab-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.3,"normalizedScore":40.3361,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-swepro-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.2,"normalizedScore":38.2353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aacodingindex-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.19,"normalizedScore":74.8551,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aascicode-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50.2,"normalizedScore":83.1366,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-lcr-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.3,"normalizedScore":96.8296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-critpt-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4,"normalizedScore":12.3839,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-designarenawebsite-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1298,"normalizedScore":85.4785,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-hle-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":48,"normalizedScore":70.7018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-hlenotools-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":34,"normalizedScore":53.5316,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aagpqadiamond-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86.6,"normalizedScore":89.3466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaomniscienceindex-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.6,"normalizedScore":71.2716,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-omniscienceaccuracy-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.6,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-omnisciencehallucinationrate-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.5,"normalizedScore":87.4548,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaopennessindex-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaifbench-2026-07-27","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":79.9,"normalizedScore":95.5752,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-tau2bench-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":97.7,"normalizedScore":98.5873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gdpvalaanormalized-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.2,"normalizedScore":42.9412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aaagenticindex-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.1,"normalizedScore":43.3352,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-apexagentsaa-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":17,"normalizedScore":35.1293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gdpvalaa-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1084,"normalizedScore":60.7576,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gertlabs-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":43.86,"normalizedScore":38.4827,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-researchclawbench-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":12.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-scicode-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":47.3,"normalizedScore":61.3293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aacodingindex-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.25,"normalizedScore":49.4982,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aascicode-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.3,"normalizedScore":78.2462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-lcr-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.3,"normalizedScore":84.9406,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-critpt-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":8,"normalizedScore":24.7678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-mmmupro-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.1,"normalizedScore":48.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-designarenawebsite-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1225,"normalizedScore":73.4323,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aammmupro-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.1,"normalizedScore":88.6598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gpqa-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.1,"normalizedScore":92.2956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-hle-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":35,"normalizedScore":47.8947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-omniscienceaccuracy-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.6,"normalizedScore":53.9519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-omnisciencehallucinationrate-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25,"normalizedScore":86.8516,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aagpqadiamond-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":90.1,"normalizedScore":94.3182,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aaomniscienceindex-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.3,"normalizedScore":82.81,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-ifbench-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":81.3,"normalizedScore":92.0601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aaifbench-2026-07-27","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":81.3,"normalizedScore":97.6401,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-bigcodebench-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"bigcodebench","benchmarkName":"BigCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"BigCodeBench authors","benchmarkVersion":"2026","score":59.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-humaneval-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-humaneval","benchmarkName":"Evaluating Large Language Models Trained on Code","benchmarkCategory":"coding","benchmarkOrganisation":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","benchmarkVersion":"2021","score":76.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-bbh-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":87.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-drop-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-drop","benchmarkName":"Discrete Reasoning Over Paragraphs","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-hellaswag-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hellaswag","benchmarkName":"HellaSwag","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-winogrande-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-winogrande","benchmarkName":"WinoGrande","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":81.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-cluewsc-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cluewsc","benchmarkName":"CLUEWSC","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-longbenchv2-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":51.5,"normalizedScore":34.5178,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-agieval-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-agieval","benchmarkName":"AGIEval","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":83.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmlu-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":90.1,"normalizedScore":85.4701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmluredux-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":90.8,"normalizedScore":78.1462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmlupro-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":73.5,"normalizedScore":77.0916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmmlu-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90.3,"normalizedScore":92,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-ceval-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":93.1,"normalizedScore":93.9394,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-cmmlu-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmmlu","benchmarkName":"Chinese Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-multiloko-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-multiloko","benchmarkName":"MultiLoKo","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":51.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-simpleqa-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":55.2,"normalizedScore":92.2414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-supergpqa-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":53.9,"normalizedScore":42.8055,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-factsparametric-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-factsparametric","benchmarkName":"FACTS Parametric","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":62.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-triviaqa-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-triviaqa","benchmarkName":"TriviaQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mgsm-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mgsm","benchmarkName":"Multilingual Grade School Math","benchmarkCategory":"knowledge","benchmarkOrganisation":"Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","benchmarkVersion":"2022","score":84.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-gsm8k-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gsm8k","benchmarkName":"Grade School Math 8K","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":92.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mathbenchmark-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mathbenchmark","benchmarkName":"MATH","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":64.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-cmath-2026-07-27","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmath","benchmarkName":"CMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-jobbench-2026-07-27","modelSlug":"claude-4-1-opus","modelName":"Claude 4.1 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":21.9,"normalizedScore":28.9582,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-sweverified-2026-07-27","modelSlug":"claude-4-1-opus","modelName":"Claude 4.1 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.5,"normalizedScore":70.3039,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-designarenawebsite-2026-07-27","modelSlug":"claude-4-1-opus","modelName":"Claude 4.1 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1207,"normalizedScore":70.462,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-pro-deep-think-arcagi2-2026-07-27","modelSlug":"gemini-3-pro-deep-think","modelName":"Gemini 3 Pro Deep Think","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":45.1,"normalizedScore":39.924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-deep-think-critpt-2026-07-27","modelSlug":"gemini-3-pro-deep-think","modelName":"Gemini 3 Pro Deep Think","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":25.7,"normalizedScore":79.5666,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-muse-spark-terminalbench2-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59,"normalizedScore":41.4591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-tau2bench-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":91.5,"normalizedScore":92.331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-deepsearchqa-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":74.8,"normalizedScore":37.2671,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-cybergym-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":43.5,"normalizedScore":0.6865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-claweval-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":63.8,"normalizedScore":81.4246,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aaagenticindex-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.69,"normalizedScore":51.6821,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-gdpvalaanormalized-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.2,"normalizedScore":47.3529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-gdpvalaa-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1143,"normalizedScore":63.7374,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-sweverified-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.4,"normalizedScore":74.3094,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-swepro-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.4,"normalizedScore":25.4011,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-livecodebenchpro-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":80,"normalizedScore":84.1456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-vibecodebench-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":19.674,"normalizedScore":27.7087,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aacodingindex-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.62,"normalizedScore":72.636,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aascicode-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":51.5,"normalizedScore":85.3288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-arcagi2-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":42.5,"normalizedScore":36.6286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-lcr-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-critpt-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":11.3,"normalizedScore":34.9845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-charxiv-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":86.4,"normalizedScore":82.598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-mmmupro-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":80.4,"normalizedScore":56.129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-erqa-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":64.7,"normalizedScore":71.978,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-simplevqa-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":71.3,"normalizedScore":59.375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-screenspotpro-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":84.1,"normalizedScore":90.9953,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-zerobench-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"2026","score":33,"normalizedScore":55.5556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-medxpertqamm-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":78.4,"normalizedScore":91.1043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aammmupro-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.5,"normalizedScore":92.7835,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-gpqadiamond-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":89.5,"normalizedScore":91.4396,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-hle-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":50.4,"normalizedScore":74.9123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-hlenotools-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":42.8,"normalizedScore":69.8885,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-healthbenchhard-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":42.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-medxpertqatext-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":52.6,"normalizedScore":11.2676,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aaomniscienceindex-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.1,"normalizedScore":71.6641,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-omniscienceaccuracy-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.6,"normalizedScore":71.134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-omnisciencehallucinationrate-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.2,"normalizedScore":28.7093,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aaifbench-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.9,"normalizedScore":89.6755,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-frontiermathv2tiers13-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":39,"normalizedScore":43.8202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-frontiermathv2tier4-2026-07-27","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.6,"normalizedScore":17.5904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-claweval-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":49.2,"normalizedScore":61.0335,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-tau2bench-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":43.3,"normalizedScore":43.6932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-gertlabs-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":56.63,"normalizedScore":65.4691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-jobbench-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":11.4,"normalizedScore":6.2162,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-vibecodebench-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":20.204,"normalizedScore":28.4551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aascicode-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49.9,"normalizedScore":82.6307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-lcr-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48,"normalizedScore":63.4082,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-critpt-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aammmupro-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.6,"normalizedScore":89.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-designarenawebsite-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1226,"normalizedScore":73.5974,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aagpqadiamond-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81.2,"normalizedScore":81.6761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aahle-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.1,"normalizedScore":21.9124,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aaomniscienceindex-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-3.6,"normalizedScore":65.6201,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-omniscienceaccuracy-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.5,"normalizedScore":72.6804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.2,"normalizedScore":8.2027,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aaifbench-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":55.1,"normalizedScore":58.9971,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-frontiermathv2tiers13-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":35.64,"normalizedScore":40.0449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-frontiermathv2tier4-2026-07-27","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aaagenticindex-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.01,"normalizedScore":37.7159,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-tau2bench-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":81.9,"normalizedScore":82.6438,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-gertlabs-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":41.24,"normalizedScore":32.9459,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.4,"normalizedScore":35.8824,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-gdpvalaa-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":988,"normalizedScore":55.9091,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-vibecodebench-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":24.606,"normalizedScore":34.6549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aacodingindex-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.39,"normalizedScore":59.5901,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aascicode-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.3,"normalizedScore":71.5008,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-lcr-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75,"normalizedScore":99.0753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-critpt-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.9,"normalizedScore":15.1703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aammmupro-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.5,"normalizedScore":84.1924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-designarenawebsite-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1217,"normalizedScore":72.1122,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aagpqadiamond-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87.3,"normalizedScore":90.3409,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aahle-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":26.5,"normalizedScore":46.6135,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.6,"normalizedScore":72.8414,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.6,"normalizedScore":59.1065,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.3,"normalizedScore":55.1267,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aaifbench-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.9,"normalizedScore":85.2507,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":31.034,"normalizedScore":34.8697,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-frontiermathv2tier4-2026-07-27","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":12.5,"normalizedScore":15.0602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-flash-aaagenticindex-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":11.99,"normalizedScore":21.313,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-tau2bench-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83.9,"normalizedScore":84.662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-gdpvalaanormalized-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.9,"normalizedScore":24.8529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-gdpvalaa-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":838,"normalizedScore":48.3333,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-sweverified-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.4,"normalizedScore":68.7845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aacodingindex-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.84,"normalizedScore":60.2261,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aascicode-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":25.9,"normalizedScore":42.1585,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-lcr-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.3,"normalizedScore":41.3474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-critpt-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-gpqa-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":83.7,"normalizedScore":83.1645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-mmlupro-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":84.9,"normalizedScore":93.3125,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aagpqadiamond-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":65.6,"normalizedScore":59.517,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aahle-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":8,"normalizedScore":9.761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aaomniscienceindex-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-48.5,"normalizedScore":30.3768,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-omniscienceaccuracy-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.2,"normalizedScore":20.6186,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-omnisciencehallucinationrate-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.1,"normalizedScore":26.4174,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aaifbench-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.9,"normalizedScore":36.5782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aime2025-2026-07-27","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":94.1,"normalizedScore":94.8745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-5-claweval-2026-07-27","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.3,"normalizedScore":79.3296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-mmclawbench-2026-07-27","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-mmclawbench","benchmarkName":"MM-ClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":23.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-terminalbench2-2026-07-27","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":65.8,"normalizedScore":53.5587,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-gertlabs-2026-07-27","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":46.89,"normalizedScore":44.8859,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-researchclawbench-2026-07-27","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":16.9,"normalizedScore":51.7241,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-swepro-2026-07-27","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.1,"normalizedScore":35.2941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-videommewithsub-2026-07-27","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.7,"normalizedScore":88.4615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-charxiv-2026-07-27","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":81,"normalizedScore":69.3627,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-mmmupro-2026-07-27","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":77.9,"normalizedScore":48.0645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-designarenawebsite-2026-07-27","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1291,"normalizedScore":84.3234,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-tau2bench-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":52.3,"normalizedScore":52.775,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-gertlabs-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.66,"normalizedScore":29.6069,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-jobbench-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":18.4,"normalizedScore":21.3775,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-sweverified-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":72.7,"normalizedScore":67.8177,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aascicode-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.3,"normalizedScore":61.3828,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-lcr-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.3,"normalizedScore":58.5205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-critpt-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aammmupro-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":62.4,"normalizedScore":61.6838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-designarenawebsite-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1175,"normalizedScore":65.1815,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aagpqadiamond-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68.3,"normalizedScore":63.3523,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aahle-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4,"normalizedScore":1.7928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aaomniscienceindex-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-9.2,"normalizedScore":61.2245,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-omniscienceaccuracy-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.4,"normalizedScore":32.9897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-omnisciencehallucinationrate-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.8,"normalizedScore":67.7925,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aaifbench-2026-07-27","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":45.4,"normalizedScore":44.6903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-terminalbench2-2026-07-27","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":64.2,"normalizedScore":50.7117,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-claweval-2026-07-27","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":69.8,"normalizedScore":89.8045,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-sweverified-2026-07-27","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":75.6,"normalizedScore":71.8232,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-swepro-2026-07-27","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":50.4,"normalizedScore":20.0535,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-swemultilingual-2026-07-27","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":69.3,"normalizedScore":43.0348,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-nl2repo-2026-07-27","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":34.6,"normalizedScore":34.1014,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aaagenticindex-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.17,"normalizedScore":10.7292,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-apexagentsaa-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":12.2,"normalizedScore":24.7845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-tau2bench-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":31.3,"normalizedScore":31.5843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-gdpvalaanormalized-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.4,"normalizedScore":10.8824,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-gdpvalaa-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":649,"normalizedScore":38.7879,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-gertlabs-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":38.46,"normalizedScore":27.071,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-vibecodebench-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aacodingindex-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.69,"normalizedScore":38.8127,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aascicode-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":41.9,"normalizedScore":69.14,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-lcr-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.3,"normalizedScore":86.2616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-critpt-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-charxiv-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":73.2,"normalizedScore":50.2451,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aammmupro-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.5,"normalizedScore":84.1924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aagpqadiamond-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":82.2,"normalizedScore":83.0966,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aahle-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":16.2,"normalizedScore":26.0956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aaomniscienceindex-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-15.5,"normalizedScore":56.2794,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-omniscienceaccuracy-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":36.4,"normalizedScore":57.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.6,"normalizedScore":18.5766,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aaifbench-2026-07-27","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":77.2,"normalizedScore":91.5929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-osworld-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":"2026","score":47.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-tau2bench-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":45.3,"normalizedScore":45.7114,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaanormalized-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaa-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":465,"normalizedScore":29.4949,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-scicode-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":32,"normalizedScore":15.1057,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aascicode-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":27.8,"normalizedScore":45.3626,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aacodingindex-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.75,"normalizedScore":9.2155,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-lcr-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.7,"normalizedScore":47.1598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-critpt-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmmu-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":70.8,"normalizedScore":71.4982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlongbenchdoc-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmlongbenchdoc","benchmarkName":"MMLongBench-Doc","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":57.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-charxiv-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":76.25,"normalizedScore":57.7206,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-screenspotpro-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":57.8,"normalizedScore":28.673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-videommenosub-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-videommenosub","benchmarkName":"Video-MME without subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":72.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-ai2dtest-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-ai2dtest","benchmarkName":"AI2D test split","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-refcocoavg-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":90.5,"normalizedScore":80.7692,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aammmupro-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":53.2,"normalizedScore":45.8763,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlupro-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":77.3,"normalizedScore":82.4986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqa-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":72.2,"normalizedScore":66.757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqadiamond-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":72.2,"normalizedScore":66.757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aahle-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.3,"normalizedScore":4.3825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaomniscienceindex-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-56,"normalizedScore":24.4898,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-omniscienceaccuracy-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.8,"normalizedScore":19.9313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-omnisciencehallucinationrate-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.1,"normalizedScore":16.7672,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-ifbench-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":74.2,"normalizedScore":76.824,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaifbench-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":63.2,"normalizedScore":70.944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aime2025-2026-07-27","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":82.1,"normalizedScore":73.6656,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-jobbench-2026-07-27","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":16,"normalizedScore":16.1793,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-sweverified-2026-07-27","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.3,"normalizedScore":68.6464,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-designarenawebsite-2026-07-27","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1152,"normalizedScore":61.3861,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-frontiermathv2tiers13-2026-07-27","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":5.903,"normalizedScore":6.6326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-frontiermathv2tier4-2026-07-27","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o4-mini-high-frontiermathv2tiers13-2026-07-27","modelSlug":"o4-mini-high","modelName":"o4-mini (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":24.828,"normalizedScore":27.8966,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o4-mini-high-frontiermathv2tier4-2026-07-27","modelSlug":"o4-mini-high","modelName":"o4-mini (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":6.25,"normalizedScore":7.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-gpqa-2026-07-27","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":77.5,"normalizedScore":74.3187,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-supergpqa-2026-07-27","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":62.6,"normalizedScore":54.9123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-mmlupro-2026-07-27","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":83,"normalizedScore":90.609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-mmluprox-2026-07-27","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":79.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-terminalbench2-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":54,"normalizedScore":32.5623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-osworldverified-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":74,"normalizedScore":76.087,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-gdpvalaa-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1139,"normalizedScore":63.5354,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aaagenticindex-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.82,"normalizedScore":48.2815,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-gdpvalaanormalized-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.9,"normalizedScore":46.9118,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aabriefcaseelo-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":636,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aatau3banking-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aaautomationbench-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-swepro-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":54.2,"normalizedScore":30.2139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aacodingindex-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.32,"normalizedScore":59.4912,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aascicode-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.9,"normalizedScore":67.4536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-mrcrv2-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":72.2,"normalizedScore":57.3705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-lcr-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62,"normalizedScore":81.9022,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-critpt-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aammmupro-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":79,"normalizedScore":90.2062,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aagpqadiamond-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":83.8,"normalizedScore":85.3693,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aahle-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":17.5,"normalizedScore":28.6853,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aaomniscienceindex-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.9,"normalizedScore":73.8619,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-omniscienceaccuracy-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.3,"normalizedScore":46.5636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.5,"normalizedScore":76.5983,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-terminalbench2-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":59.5,"normalizedScore":42.3488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-browsecomp-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":75.82,"normalizedScore":65.7322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-deepsearchqa-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":92.82,"normalizedScore":93.2298,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-gdpvalaanormalized-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.8,"normalizedScore":37.9412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-toolathlon-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":49.5,"normalizedScore":46.4066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-claweval-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":67.1,"normalizedScore":86.0335,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-hlewithtools-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":47.2,"normalizedScore":35.8974,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-gertlabs-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":51.57,"normalizedScore":54.776,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aaagenticindex-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.53,"normalizedScore":38.6616,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-tau2bench-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-gdpvalaa-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1017,"normalizedScore":57.3737,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-apexagentsaa-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":14.8,"normalizedScore":30.3879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-swepro-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.3,"normalizedScore":35.8289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aacodingindex-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.57,"normalizedScore":45.7102,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aascicode-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40,"normalizedScore":65.9359,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-lcr-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.7,"normalizedScore":84.148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-critpt-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.3,"normalizedScore":7.1207,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-simplevqa-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":79.2,"normalizedScore":90.2344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-vstar-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":95.3,"normalizedScore":94.6488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aammmupro-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.3,"normalizedScore":83.8488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-designarenawebsite-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1211,"normalizedScore":71.1221,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aagpqadiamond-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":80.9,"normalizedScore":81.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aahle-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":19.9,"normalizedScore":33.4661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aaomniscienceindex-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-37.5,"normalizedScore":39.011,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-omniscienceaccuracy-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.4,"normalizedScore":38.1443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-omnisciencehallucinationrate-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.4,"normalizedScore":15.199,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aaifbench-2026-07-27","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":67.3,"normalizedScore":76.9912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-tau2bench-2026-07-27","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":28.7,"normalizedScore":28.9606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-sweverified-2026-07-27","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":49.3,"normalizedScore":35.4972,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aascicode-2026-07-27","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.9,"normalizedScore":65.7673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-mmlu-2026-07-27","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":86.9,"normalizedScore":58.1197,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-gpqa-2026-07-27","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":77.2,"normalizedScore":73.8907,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aagpqadiamond-2026-07-27","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":74.8,"normalizedScore":72.5852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aahle-2026-07-27","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":8.7,"normalizedScore":11.1554,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-ifeval-2026-07-27","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":93.9,"normalizedScore":96.7494,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aime2024-2026-07-27","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aime2024","benchmarkName":"American Invitational Mathematics Examination 2024","benchmarkCategory":"mathematics","benchmarkOrganisation":"Mathematical Association of America","benchmarkVersion":"2024","score":87.3,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-bfclv4-2026-07-27","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":45.6,"normalizedScore":45.5253,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-mmluredux-2026-07-27","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":60.8139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-gpqa-2026-07-27","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":57.6,"normalizedScore":45.9267,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-gpqadiamond-2026-07-27","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":57.6,"normalizedScore":45.9267,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-ifeval-2026-07-27","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":76.5,"normalizedScore":45.331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-bfclv4-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":49.73,"normalizedScore":53.1777,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-tau2bench-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":16.1,"normalizedScore":16.2462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aascicode-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":7.8,"normalizedScore":11.6358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-lcr-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-critpt-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aagpqadiamond-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":46.6,"normalizedScore":32.5284,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aahle-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.9,"normalizedScore":7.5697,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aaomniscienceindex-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-33.3,"normalizedScore":42.3077,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-omniscienceaccuracy-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.4,"normalizedScore":10.6529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-omnisciencehallucinationrate-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47,"normalizedScore":60.3136,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-ifeval-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":91.84,"normalizedScore":90.6619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-ifbench-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":56.47,"normalizedScore":38.7768,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aaifbench-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":53.3,"normalizedScore":56.3422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-math500-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-math500","benchmarkName":"MATH-500 Problem Set","benchmarkCategory":"mathematics","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2021","score":88.76,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aime2025-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":42.53,"normalizedScore":3.7292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aime2026-2026-07-27","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":50,"normalizedScore":16.2981,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-jobbench-2026-07-27","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":18.5,"normalizedScore":21.5941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-vibecodebench-2026-07-27","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":15.738,"normalizedScore":22.1653,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-frontiermathv2tiers13-2026-07-27","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":21.034,"normalizedScore":23.6337,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-frontiermathv2tier4-2026-07-27","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-terminalbench2-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":57,"normalizedScore":37.9004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-tau2bench-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-toolathlon-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":46.3,"normalizedScore":39.8357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-mlebenchlite-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mlebenchlite","benchmarkName":"MLE-Bench Lite","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":66.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-mmclawbench-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mmclawbench","benchmarkName":"MM-ClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":62.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-claweval-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":48.7,"normalizedScore":60.3352,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aaagenticindex-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.58,"normalizedScore":46.0266,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-apexagentsaa-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":10.6,"normalizedScore":21.3362,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gdpvalaanormalized-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.9,"normalizedScore":48.3824,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gdpvalaa-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1159,"normalizedScore":64.5455,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gertlabs-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":40.4,"normalizedScore":31.1708,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-sweverifiedarcee-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75.4,"normalizedScore":98.3871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-swepro-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.2,"normalizedScore":35.5615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-swerebench-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":51.9,"normalizedScore":43.4599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-swemultilingual-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":76.5,"normalizedScore":60.9453,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-multiswebench-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"multi-swe-bench","benchmarkName":"Multi-SWE-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"Multi-SWE-Bench","benchmarkVersion":"2026","score":52.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-vibepro-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibepro","benchmarkName":"VIBE-Pro","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":55.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-nl2repo-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":39.8,"normalizedScore":58.0645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-vibecodebench-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":27.037,"normalizedScore":38.0787,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-reactnativeevals-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71.4,"normalizedScore":1.5936,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aacodingindex-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.62,"normalizedScore":64.1555,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aascicode-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47,"normalizedScore":77.7403,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-lcr-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.7,"normalizedScore":90.753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-critpt-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-designarenawebsite-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1275,"normalizedScore":81.6832,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gpqadiamond-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-mmluproarcee-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":80.8,"normalizedScore":40.2878,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aahle-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.1,"normalizedScore":49.8008,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aaomniscienceindex-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.7,"normalizedScore":68.9953,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-omniscienceaccuracy-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.1,"normalizedScore":39.3471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-omnisciencehallucinationrate-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.4,"normalizedScore":75.5127,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aaifbench-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.7,"normalizedScore":89.3805,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aime2025arcee-2026-07-27","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":80,"normalizedScore":73.8786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-tau2bench-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":61.1,"normalizedScore":61.6549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aascicode-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":34.5,"normalizedScore":56.661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-lcr-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51,"normalizedScore":67.3712,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-critpt-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-designarenawebsite-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1080,"normalizedScore":49.505,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aagpqadiamond-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.6,"normalizedScore":75.142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aahle-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7,"normalizedScore":7.7689,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aaomniscienceindex-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-27.5,"normalizedScore":46.8603,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-omniscienceaccuracy-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":40.5498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-omnisciencehallucinationrate-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.2,"normalizedScore":27.503,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aaifbench-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":41.5,"normalizedScore":38.9381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-frontiermathv2tiers13-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":21.404,"normalizedScore":24.0494,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-frontiermathv2tier4-2026-07-27","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-tau2bench-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74.9,"normalizedScore":75.5802,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-gertlabs-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":42.34,"normalizedScore":35.2705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-reactnativeevals-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":72.6,"normalizedScore":6.3745,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aascicode-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.7,"normalizedScore":75.5481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-lcr-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68,"normalizedScore":89.8283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-critpt-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2,"normalizedScore":6.192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aammmupro-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":68.8,"normalizedScore":72.6804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aagpqadiamond-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87.7,"normalizedScore":90.9091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aahle-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.9,"normalizedScore":41.4343,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aaomniscienceindex-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.8,"normalizedScore":71.4286,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-omniscienceaccuracy-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.4,"normalizedScore":65.6357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-omnisciencehallucinationrate-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.2,"normalizedScore":39.5657,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aaifbench-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":53.7,"normalizedScore":56.9322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-frontiermathv2tiers13-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":19.655,"normalizedScore":22.0843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-frontiermathv2tier4-2026-07-27","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-humaneval-2026-07-27","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-humaneval","benchmarkName":"Evaluating Large Language Models Trained on Code","benchmarkCategory":"coding","benchmarkOrganisation":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","benchmarkVersion":"2021","score":73.8,"normalizedScore":58.9041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-bbh-2026-07-27","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":78.8,"normalizedScore":74.7826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-drop-2026-07-27","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-drop","benchmarkName":"Discrete Reasoning Over Paragraphs","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":66.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-gpqa-2026-07-27","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":43.4,"normalizedScore":25.667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-gpqadiamond-2026-07-27","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":43.4,"normalizedScore":25.667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-mmlupro-2026-07-27","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":51.4,"normalizedScore":45.646,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-agieval-2026-07-27","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-agieval","benchmarkName":"AGIEval","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":66.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-gsm8k-2026-07-27","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-gsm8k","benchmarkName":"Grade School Math 8K","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":86.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-tau2bench-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":80.7,"normalizedScore":81.4329,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aascicode-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":41,"normalizedScore":67.6223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-lcr-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":91.5456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-critpt-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aammmupro-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":70.1,"normalizedScore":74.9141,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-designarenawebsite-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1067,"normalizedScore":47.3597,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aagpqadiamond-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":82.7,"normalizedScore":83.8068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aahle-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":20,"normalizedScore":33.6653,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aaomniscienceindex-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-15.3,"normalizedScore":56.4364,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-omniscienceaccuracy-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.4,"normalizedScore":60.4811,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-omnisciencehallucinationrate-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.1,"normalizedScore":11.9421,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aaifbench-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":71.4,"normalizedScore":83.0383,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aamath500-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aamath500","benchmarkName":"Artificial Analysis MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99.2,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-frontiermathv2tiers13-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":18.685,"normalizedScore":20.9944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-frontiermathv2tier4-2026-07-27","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aaagenticindex-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.97,"normalizedScore":19.4581,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-tau2bench-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":43.6,"normalizedScore":43.996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-gdpvalaanormalized-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.5,"normalizedScore":19.8529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-gdpvalaa-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":770,"normalizedScore":44.899,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aacodingindex-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.32,"normalizedScore":45.3569,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aascicode-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40,"normalizedScore":65.9359,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-lcr-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.7,"normalizedScore":73.5799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-critpt-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-mmmupro-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":73.8,"normalizedScore":34.8387,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aammmupro-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":69.2,"normalizedScore":73.3677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-mmlupro-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":82.6,"normalizedScore":90.0398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-hle-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":17.2,"normalizedScore":16.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-hlenotools-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":8.7,"normalizedScore":6.5056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aagpqadiamond-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":79.2,"normalizedScore":78.8352,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aaomniscienceindex-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-48.1,"normalizedScore":30.6907,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-omniscienceaccuracy-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.2,"normalizedScore":25.7732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.9,"normalizedScore":19.421,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aaifbench-2026-07-27","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.4,"normalizedScore":84.5133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-terminalbench2-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":54.4,"normalizedScore":33.274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gertlabs-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":36.91,"normalizedScore":23.7954,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aaagenticindex-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.73,"normalizedScore":55.3919,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gdpvalaanormalized-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.8,"normalizedScore":52.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gdpvalaa-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1215,"normalizedScore":67.3737,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-sweverified-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.4,"normalizedScore":70.1657,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-scicode-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":41.2,"normalizedScore":42.9003,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aascicode-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.6,"normalizedScore":78.7521,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aacodingindex-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.8,"normalizedScore":72.8905,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-lcr-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-critpt-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.9,"normalizedScore":15.1703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gpqa-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.2,"normalizedScore":88.1581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gpqadiamond-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87.2,"normalizedScore":88.1581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-hle-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":25.5,"normalizedScore":31.2281,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-omniscienceaccuracy-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.5,"normalizedScore":48.6254,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-omnisciencehallucinationrate-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73,"normalizedScore":28.9505,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aaomniscienceindex-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-18.5,"normalizedScore":53.9246,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-ifbench-2026-07-27","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":63.1,"normalizedScore":53.0043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aaagenticindex-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.17,"normalizedScore":1.6367,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-tau2bench-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":17.3,"normalizedScore":17.4571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-gdpvalaa-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":63,"normalizedScore":9.1919,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aacodingindex-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":11.14,"normalizedScore":5.5265,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aascicode-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":25.9,"normalizedScore":42.1585,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-lcr-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17,"normalizedScore":22.4571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-critpt-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aammmupro-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":40.1,"normalizedScore":23.3677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-designarenawebsite-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1003,"normalizedScore":36.7987,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-mmlu-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":80.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-gpqa-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":50.3,"normalizedScore":35.5115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aagpqadiamond-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":51.2,"normalizedScore":39.0625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aahle-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.9,"normalizedScore":1.5936,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aaomniscienceindex-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-56.4,"normalizedScore":24.1758,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.3,"normalizedScore":17.354,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.4,"normalizedScore":20.0241,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-ifeval-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":83.2,"normalizedScore":65.13,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aaifbench-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":32,"normalizedScore":24.9263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":1.034,"normalizedScore":1.1618,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-tau2airline-2026-07-27","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-tau2airline","benchmarkName":"τ²-Bench Airline Domain","benchmarkCategory":"agents","benchmarkOrganisation":"Victor Barres, Honghua Dong, Soham Ray, Xujie Si, Karthik Narasimhan","benchmarkVersion":"2025","score":56.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-livecodebenchv6-2026-07-27","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":65.7,"normalizedScore":53.9209,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-sweverified-2026-07-27","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":53.2,"normalizedScore":40.884,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-mmlupro-2026-07-27","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":68.1,"normalizedScore":69.4081,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-gpqa-2026-07-27","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":57.3,"normalizedScore":45.4986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-gpqadiamond-2026-07-27","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":57.3,"normalizedScore":45.4986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-aime2026-2026-07-27","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":76.4,"normalizedScore":61.2113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-terminalbench2-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":50,"normalizedScore":25.4448,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-osworldverified-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":61.4,"normalizedScore":48.6957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-vitabench-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":17,"normalizedScore":4.6296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-gertlabs-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":48.51,"normalizedScore":48.3094,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-jobbench-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":27.7,"normalizedScore":41.5205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-sweverified-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.2,"normalizedScore":74.0331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-arcagi2-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":13.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-designarenawebsite-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1219,"normalizedScore":72.4422,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-gpqa-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":83.4,"normalizedScore":82.7365,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-aime2025-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":87,"normalizedScore":82.3259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-frontiermathv2tiers13-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":13.495,"normalizedScore":15.1629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-frontiermathv2tier4-2026-07-27","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaagenticindex-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.45,"normalizedScore":25.7865,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-tau2bench-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":59.9,"normalizedScore":60.444,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gdpvalaanormalized-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.5,"normalizedScore":22.7941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gdpvalaa-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":811,"normalizedScore":46.9697,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gertlabs-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":35.26,"normalizedScore":20.3085,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaenterpriseopsgym-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaitbench-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.3,"normalizedScore":62.6482,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-terminalbenchhard-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":36.4,"normalizedScore":27.8729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-swerebench-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":41.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-reactnativeevals-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":75.2,"normalizedScore":16.7331,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aacodingindex-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.43,"normalizedScore":51.1661,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aascicode-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.4,"normalizedScore":71.6695,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-lcr-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62,"normalizedScore":81.9022,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-critpt-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-mmmupro-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":76.9,"normalizedScore":44.8387,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aammmupro-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":73.4,"normalizedScore":80.5842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gpqa-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":84.3,"normalizedScore":84.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-mmlupro-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.2,"normalizedScore":93.7393,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-hle-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":26.5,"normalizedScore":32.9825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-hlenotools-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":19.5,"normalizedScore":26.5799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aagpqadiamond-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.7,"normalizedScore":88.0682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaomniscienceindex-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-45.4,"normalizedScore":32.81,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-omniscienceaccuracy-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.9,"normalizedScore":28.6942,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.6,"normalizedScore":18.5766,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaopennessindex-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaifbench-2026-07-27","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.6,"normalizedScore":89.233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-tau2bench-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":47.1,"normalizedScore":47.5277,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-gertlabs-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":25.65,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-sweverified-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":54.6,"normalizedScore":42.8177,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aascicode-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.1,"normalizedScore":62.7319,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-lcr-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61,"normalizedScore":80.5812,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-critpt-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aammmupro-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":61.2,"normalizedScore":59.622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-designarenawebsite-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1068,"normalizedScore":47.5248,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mmlu-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":90.2,"normalizedScore":86.3248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-gpqa-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":66.3,"normalizedScore":58.3393,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aagpqadiamond-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":66.6,"normalizedScore":60.9375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aahle-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aaomniscienceindex-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-36.2,"normalizedScore":40.0314,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.2,"normalizedScore":36.0825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":79.6,"normalizedScore":20.9891,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-ifeval-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":87.4,"normalizedScore":77.5414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aaifbench-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43,"normalizedScore":41.1504,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":5.517,"normalizedScore":6.1989,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-frontiermathv2tier4-2026-07-27","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tiers13-2026-07-27","modelSlug":"qwen3-235b-2507-reasoning","modelName":"Qwen3 235B 2507 (Reasoning)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":8.481,"normalizedScore":9.5292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tier4-2026-07-27","modelSlug":"qwen3-235b-2507-reasoning","modelName":"Qwen3 235B 2507 (Reasoning)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-tau2bench-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":36.3,"normalizedScore":36.6297,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aaagenticindex-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.93,"normalizedScore":13.9298,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-gdpvalaanormalized-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.6,"normalizedScore":11.1765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-gdpvalaa-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":651,"normalizedScore":38.8889,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aascicode-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.2,"normalizedScore":62.9005,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aacodingindex-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.96,"normalizedScore":33.5406,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-bbh-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":53,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mrcrv2-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":43.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-lcr-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.3,"normalizedScore":73.0515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-critpt-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mmmupro-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":69.1,"normalizedScore":19.6774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mathvision-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":79.7,"normalizedScore":27,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-medxpertqamm-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":48.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aammmupro-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":69.7,"normalizedScore":74.2268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-gpqa-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":78.8,"normalizedScore":76.1735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-gpqadiamond-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":78.8,"normalizedScore":76.1735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mmlupro-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":77.2,"normalizedScore":82.3563,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-hlenotools-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":5.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mmmlu-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aahle-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.8,"normalizedScore":23.3068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aaomniscienceindex-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-51.9,"normalizedScore":27.708,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-omniscienceaccuracy-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16,"normalizedScore":21.9931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.8,"normalizedScore":19.5416,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aaifbench-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.5,"normalizedScore":86.1357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aime2026-2026-07-27","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":77.5,"normalizedScore":63.0827,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-terminalbench2-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":47.1,"normalizedScore":20.2847,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-deepsearchqa-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":62.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-gertlabs-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":38.36,"normalizedScore":26.8597,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-livecodebenchpro-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":74.2,"normalizedScore":75.6312,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-sweverified-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":76.7,"normalizedScore":73.3425,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-swepro-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":51.8,"normalizedScore":23.7968,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-vibecodebench-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":4.064,"normalizedScore":5.7237,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-arcagi2-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":53.3,"normalizedScore":50.3169,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-arcagi3-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.09,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-mmmupro-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":75.2,"normalizedScore":39.3548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-charxiv-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":60.9,"normalizedScore":20.098,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-erqa-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":54.1,"normalizedScore":13.7363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-simplevqa-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":57.4,"normalizedScore":5.0781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-medxpertqamm-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":65.8,"normalizedScore":52.454,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-designarenawebsite-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1257,"normalizedScore":78.7129,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-gpqadiamond-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":88.5,"normalizedScore":90.0128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-hlenotools-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":31.6,"normalizedScore":49.0706,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-healthbenchhard-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":20.3,"normalizedScore":19.6429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-medxpertqatext-2026-07-27","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":50.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aaagenticindex-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.72,"normalizedScore":2.6368,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-tau2bench-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":52.9,"normalizedScore":53.3804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.4,"normalizedScore":0.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-gdpvalaa-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":508,"normalizedScore":31.6667,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-sweverified-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":23.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aacodingindex-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.21,"normalizedScore":18.3463,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aascicode-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.4,"normalizedScore":66.6105,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-lcr-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.3,"normalizedScore":55.8785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-critpt-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aammmupro-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":58.7,"normalizedScore":55.3265,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-designarenawebsite-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1027,"normalizedScore":40.7591,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-mmlu-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":87.5,"normalizedScore":63.2479,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-gpqa-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":64.2,"normalizedScore":55.3431,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aagpqadiamond-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":66.4,"normalizedScore":60.6534,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aahle-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aaomniscienceindex-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-50.1,"normalizedScore":29.1209,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.5,"normalizedScore":24.5704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82,"normalizedScore":18.0941,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-ifeval-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":88.5,"normalizedScore":80.792,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aaifbench-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":38.3,"normalizedScore":34.2183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.483,"normalizedScore":5.0371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-tau2bench-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":14.9,"normalizedScore":15.0353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aascicode-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.1,"normalizedScore":47.5548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-lcr-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.9,"normalizedScore":60.6341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-critpt-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aammmupro-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":65.5,"normalizedScore":67.0103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-designarenawebsite-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1145,"normalizedScore":60.231,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aagpqadiamond-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68.3,"normalizedScore":63.3523,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aahle-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.1,"normalizedScore":3.9841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aaomniscienceindex-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-42,"normalizedScore":35.4788,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-omniscienceaccuracy-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.5,"normalizedScore":40.0344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.3,"normalizedScore":4.4632,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aaifbench-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39,"normalizedScore":35.2507,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-frontiermathv2tiers13-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.844,"normalizedScore":5.4427,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-frontiermathv2tier4-2026-07-27","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-flash-frontiermathv2tiers13-2026-07-27","modelSlug":"qwen3-5-flash","modelName":"Qwen3.5 Flash","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":6.207,"normalizedScore":6.9742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-flash-frontiermathv2tier4-2026-07-27","modelSlug":"qwen3-5-flash","modelName":"Qwen3.5 Flash","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-tau2bench-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":76.9,"normalizedScore":77.5984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-vibecodebench-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":3.09,"normalizedScore":4.3519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aascicode-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.1,"normalizedScore":54.3002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-lcr-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.3,"normalizedScore":34.7424,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-critpt-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aagpqadiamond-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":63.2,"normalizedScore":56.108,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aahle-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.2,"normalizedScore":4.1833,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aaomniscienceindex-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-31.6,"normalizedScore":43.6421,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-omniscienceaccuracy-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.8,"normalizedScore":30.2405,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-omnisciencehallucinationrate-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.1,"normalizedScore":37.2738,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aaifbench-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.7,"normalizedScore":31.8584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-frontiermathv2tiers13-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":3.819,"normalizedScore":4.291,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-frontiermathv2tier4-2026-07-27","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.128,"normalizedScore":2.5639,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-3-beta-frontiermathv2tiers13-2026-07-27","modelSlug":"grok-3-beta","modelName":"Grok 3 [Beta]","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":3.793,"normalizedScore":4.2618,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-3-beta-frontiermathv2tier4-2026-07-27","modelSlug":"grok-3-beta","modelName":"Grok 3 [Beta]","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-bigcodebench-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"bigcodebench","benchmarkName":"BigCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"BigCodeBench authors","benchmarkVersion":"2026","score":56.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-humaneval-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-humaneval","benchmarkName":"Evaluating Large Language Models Trained on Code","benchmarkCategory":"coding","benchmarkOrganisation":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","benchmarkVersion":"2021","score":69.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-bbh-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":86.9,"normalizedScore":98.2609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-drop-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-drop","benchmarkName":"Discrete Reasoning Over Paragraphs","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88.6,"normalizedScore":99.5495,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-hellaswag-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hellaswag","benchmarkName":"HellaSwag","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-winogrande-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-winogrande","benchmarkName":"WinoGrande","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":79.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-cluewsc-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cluewsc","benchmarkName":"CLUEWSC","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":82.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-longbenchv2-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":44.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-agieval-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-agieval","benchmarkName":"AGIEval","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":82.6,"normalizedScore":96.9136,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmlu-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":88.7,"normalizedScore":73.5043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmluredux-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":89.4,"normalizedScore":72.8711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmlupro-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":68.3,"normalizedScore":69.6927,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmmlu-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":88.8,"normalizedScore":72,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-ceval-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":92.1,"normalizedScore":63.6364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-cmmlu-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmmlu","benchmarkName":"Chinese Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-multiloko-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-multiloko","benchmarkName":"MultiLoKo","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":42.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-simpleqa-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":30.1,"normalizedScore":20.1149,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-supergpqa-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":46.5,"normalizedScore":32.5077,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-factsparametric-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-factsparametric","benchmarkName":"FACTS Parametric","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":33.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-triviaqa-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-triviaqa","benchmarkName":"TriviaQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":82.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mgsm-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mgsm","benchmarkName":"Multilingual Grade School Math","benchmarkCategory":"knowledge","benchmarkOrganisation":"Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","benchmarkVersion":"2022","score":85.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-gsm8k-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gsm8k","benchmarkName":"Grade School Math 8K","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.8,"normalizedScore":72.3077,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mathbenchmark-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mathbenchmark","benchmarkName":"MATH","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":57.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-cmath-2026-07-27","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmath","benchmarkName":"CMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":93.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-terminalbench2-2026-07-27","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":45.8,"normalizedScore":17.9715,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-sweverified-2026-07-27","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.6,"normalizedScore":70.442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-swemultilingual-2026-07-27","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":63.1,"normalizedScore":27.6119,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-swepro-2026-07-27","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":49.2,"normalizedScore":16.8449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aaagenticindex-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.58,"normalizedScore":2.3823,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-tau2bench-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":22.8,"normalizedScore":23.0071,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-gdpvalaanormalized-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-gdpvalaa-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":231,"normalizedScore":17.6768,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-livecodebench-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":37.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-sweverified-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":42,"normalizedScore":25.4144,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aacodingindex-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.04,"normalizedScore":22.3463,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aascicode-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.4,"normalizedScore":58.1788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-lcr-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29,"normalizedScore":38.3091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-critpt-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-designarenawebsite-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1150,"normalizedScore":61.0561,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-gpqa-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":59.1,"normalizedScore":48.0668,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-mmlupro-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":75.9,"normalizedScore":80.5065,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aagpqadiamond-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":55.7,"normalizedScore":45.4545,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aahle-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.6,"normalizedScore":0.996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aaomniscienceindex-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-41.3,"normalizedScore":36.0283,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-omniscienceaccuracy-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.4,"normalizedScore":38.1443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-omnisciencehallucinationrate-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.4,"normalizedScore":9.1677,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-ifeval-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":86.1,"normalizedScore":73.6998,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aaifbench-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":34.8,"normalizedScore":29.056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-frontiermathv2tiers13-2026-07-27","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":1.724,"normalizedScore":1.9371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-bfclv4-2026-07-27","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":44.2,"normalizedScore":42.9313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-mmluredux-2026-07-27","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":78.1,"normalizedScore":30.2939,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-gpqa-2026-07-27","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":40.9,"normalizedScore":22.1002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-gpqadiamond-2026-07-27","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":40.9,"normalizedScore":22.1002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-ifeval-2026-07-27","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":75.8,"normalizedScore":43.2624,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-sweverified-2026-07-27","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":49,"normalizedScore":35.0829,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-gpqa-2026-07-27","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":59.4,"normalizedScore":48.4948,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-frontiermathv2tiers13-2026-07-27","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.069,"normalizedScore":2.3247,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-frontiermathv2tier4-2026-07-27","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-tau2bench-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":25.1,"normalizedScore":25.328,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aascicode-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.3,"normalizedScore":54.6374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-lcr-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-critpt-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-designarenawebsite-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":861,"normalizedScore":13.3663,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aagpqadiamond-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":54.3,"normalizedScore":43.4659,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aahle-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.3,"normalizedScore":0.3984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aaomniscienceindex-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.7,"normalizedScore":60.0471,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.7,"normalizedScore":28.3505,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.9,"normalizedScore":71.2907,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aaifbench-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":34.3,"normalizedScore":28.3186,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-frontiermathv2tiers13-2026-07-27","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0.345,"normalizedScore":0.3876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aaagenticindex-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.1,"normalizedScore":1.5094,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-tau2bench-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":15.5,"normalizedScore":15.6408,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-gdpvalaanormalized-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-gdpvalaa-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":111,"normalizedScore":11.6162,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aacodingindex-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.17,"normalizedScore":1.3286,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aascicode-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":17,"normalizedScore":27.1501,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-lcr-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.8,"normalizedScore":34.0819,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-critpt-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aammmupro-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":52.9,"normalizedScore":45.3608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-designarenawebsite-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":780,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aagpqadiamond-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":58.7,"normalizedScore":49.7159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aahle-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.3,"normalizedScore":2.3904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aaomniscienceindex-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-52.4,"normalizedScore":27.3155,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-omniscienceaccuracy-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.6,"normalizedScore":19.5876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-omnisciencehallucinationrate-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":78.3,"normalizedScore":22.5573,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aaifbench-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.5,"normalizedScore":35.9882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-frontiermathv2tiers13-2026-07-27","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aaagenticindex-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.31,"normalizedScore":1.8913,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-tau2bench-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":17.8,"normalizedScore":17.9617,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-gdpvalaanormalized-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-gdpvalaa-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":7,"normalizedScore":6.3636,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aacodingindex-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.28,"normalizedScore":12.7915,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aascicode-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.1,"normalizedScore":54.3002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-lcr-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46,"normalizedScore":60.7662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-critpt-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aammmupro-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":62.1,"normalizedScore":61.1684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-designarenawebsite-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":901,"normalizedScore":19.967,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aagpqadiamond-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":67.1,"normalizedScore":61.6477,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aahle-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.8,"normalizedScore":3.3865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aaomniscienceindex-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-41.8,"normalizedScore":35.6358,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-omniscienceaccuracy-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.3,"normalizedScore":36.2543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-omnisciencehallucinationrate-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.3,"normalizedScore":11.7008,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aaifbench-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43,"normalizedScore":41.1504,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-frontiermathv2tiers13-2026-07-27","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0.69,"normalizedScore":0.7753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-tau2bench-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86,"normalizedScore":86.781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-gdpvalaanormalized-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.5,"normalizedScore":3.6765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-gdpvalaa-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":550,"normalizedScore":33.7879,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aaagenticindex-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.25,"normalizedScore":3.6007,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-scicode-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":27,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aacodingindex-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.26,"normalizedScore":25.4841,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aascicode-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":27.1,"normalizedScore":44.1821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-lcr-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25,"normalizedScore":33.0251,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-critpt-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-gpqa-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":59,"normalizedScore":47.9241,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aagpqadiamond-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":59.3,"normalizedScore":50.5682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aahle-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.2,"normalizedScore":6.1753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aaomniscienceindex-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-65.7,"normalizedScore":16.876,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-omniscienceaccuracy-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.4,"normalizedScore":20.9622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-omnisciencehallucinationrate-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":95.8,"normalizedScore":1.4475,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-ifbench-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":57,"normalizedScore":39.9142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aaifbench-2026-07-27","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":57.4,"normalizedScore":62.3894,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aaagenticindex-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.1,"normalizedScore":12.4204,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-tau2bench-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":54.1,"normalizedScore":54.5913,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gertlabs-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":42.01,"normalizedScore":34.5731,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gdpvalaanormalized-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.6,"normalizedScore":12.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gdpvalaa-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":672,"normalizedScore":39.9495,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-sweverified-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":63.8,"normalizedScore":55.5249,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-vibecodebench-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":0.4,"normalizedScore":0.5634,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aacodingindex-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.25,"normalizedScore":36.7774,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aascicode-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42.8,"normalizedScore":70.6577,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-lcr-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66,"normalizedScore":87.1863,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-critpt-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.6,"normalizedScore":8.0495,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aammmupro-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.9,"normalizedScore":83.1615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-designarenawebsite-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1197,"normalizedScore":68.8119,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gpqa-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":83,"normalizedScore":82.1658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-hle-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":18.8,"normalizedScore":19.4737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aagpqadiamond-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.4,"normalizedScore":86.2216,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aaomniscienceindex-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-14.3,"normalizedScore":57.2214,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-omniscienceaccuracy-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39,"normalizedScore":61.512,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.4,"normalizedScore":11.5802,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aaifbench-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":48.7,"normalizedScore":49.5575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-frontiermathv2tiers13-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.138,"normalizedScore":15.8854,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-frontiermathv2tier4-2026-07-27","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-terminalbench2-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":49.1,"normalizedScore":23.8434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-mcpatlas-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":64,"normalizedScore":58.8737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-toolathlon-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":40.7,"normalizedScore":28.3368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-claweval-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.8,"normalizedScore":73.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-gertlabs-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":54.35,"normalizedScore":60.6509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-sweverified-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.7,"normalizedScore":69.1989,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-swepro-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":49.1,"normalizedScore":16.5775,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-swemultilingual-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":69.7,"normalizedScore":44.0299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-mrcr1m-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":37.5,"normalizedScore":19.1564,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-corpusqa1m-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":15.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-designarenawebsite-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1238,"normalizedScore":75.5776,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-mmlupro-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":83,"normalizedScore":90.609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-simpleqa-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":23.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-chinesesimpleqa-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":71.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-gpqa-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":71.2,"normalizedScore":65.3303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-gpqadiamond-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":71.2,"normalizedScore":65.3303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-hle-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":8.1,"normalizedScore":0.7018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-hmmtfeb2026-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":40.8,"normalizedScore":21.0821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-imoanswerbench-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":41.9,"normalizedScore":12.0658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-apex-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":1,"normalizedScore":1.3605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-apexshortlist-2026-07-27","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":9.3,"normalizedScore":0.1235,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-bfclv4-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":25.15,"normalizedScore":7.6339,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-livecodebenchpro-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":22.68,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-livecodebenchv6-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":33.52,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-bbh-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":71.89,"normalizedScore":54.7536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-mmlupro-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":48.85,"normalizedScore":42.0176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-mmluredux-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":70.06,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-gpqadiamond-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":26.26,"normalizedScore":1.2127,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-supergpqa-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":23.14,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-ifbench-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":46.67,"normalizedScore":17.7468,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-ifeval-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":80.41,"normalizedScore":56.8853,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-aime2025-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":40.42,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-aime2026-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":40.42,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-hmmtfeb2026-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":25.76,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-math500-2026-07-27","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-math500","benchmarkName":"MATH-500 Problem Set","benchmarkCategory":"mathematics","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2021","score":91.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-terminalbench2-2026-07-27","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":43.1,"normalizedScore":13.1673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-claweval-2026-07-27","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":63.1,"normalizedScore":80.4469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-sweverified-2026-07-27","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":69.4,"normalizedScore":63.2597,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-swepro-2026-07-27","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":42.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-swemultilingual-2026-07-27","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":52,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-nl2repo-2026-07-27","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":27.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-terminalbench2-2026-07-27","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2026","score":35.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-sweverified-2026-07-27","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":69.9,"normalizedScore":63.9503,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-swemultilingual-2026-07-27","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":57.7,"normalizedScore":14.1791,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-swepro-2026-07-27","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":46.3,"normalizedScore":9.0909,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-bfclv4-2026-07-27","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":21.08,"normalizedScore":0.0926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-mmmu-2026-07-27","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":32.67,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-realworldqa-2026-07-27","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":58.43,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-countbench-2026-07-27","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-countbench","benchmarkName":"CountBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":73.31,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-gpqa-2026-07-27","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":25.66,"normalizedScore":0.3567,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-mmlupro-2026-07-27","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":19.32,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-ifeval-2026-07-27","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":61.16,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-tau2bench-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":80.7,"normalizedScore":81.4329,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaagenticindex-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.16,"normalizedScore":16.1666,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-gdpvalaanormalized-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.9,"normalizedScore":16.0294,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-gdpvalaa-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":718,"normalizedScore":42.2727,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-terminalbenchhard-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":25,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aacodingindex-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.85,"normalizedScore":29.1449,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aascicode-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.8,"normalizedScore":62.226,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-lcr-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46,"normalizedScore":60.7662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-critpt-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-mmmu-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":75.1,"normalizedScore":79.5612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-mmmupro-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":63,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-charxiv-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":52.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aammmupro-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":63.2,"normalizedScore":63.0584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aagpqadiamond-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.1,"normalizedScore":74.4318,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aahle-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":11.4,"normalizedScore":16.5339,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaomniscienceindex-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-4,"normalizedScore":65.3061,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-omniscienceaccuracy-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.9,"normalizedScore":9.7938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-omnisciencehallucinationrate-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.1,"normalizedScore":100,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaopennessindex-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaifbench-2026-07-27","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.9,"normalizedScore":86.7257,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-bfclv4-2026-07-27","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":21.03,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-gpqa-2026-07-27","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":25.41,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-gpqadiamond-2026-07-27","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":25.41,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-mmlupro-2026-07-27","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":20.25,"normalizedScore":1.3233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-ifeval-2026-07-27","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":71.71,"normalizedScore":31.1761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-ifbench-2026-07-27","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":38.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-researchclawbench-2026-07-27","modelSlug":"grok-4-1","modelName":"Grok 4.1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":13.5,"normalizedScore":12.6437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-reasoning-vibecodebench-2026-07-27","modelSlug":"glm-5-reasoning","modelName":"GLM-5 (Reasoning)","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":23.359,"normalizedScore":32.8986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-reasoning-designarenawebsite-2026-07-27","modelSlug":"glm-5-reasoning","modelName":"GLM-5 (Reasoning)","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1278,"normalizedScore":82.1782,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-claweval-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":55.8,"normalizedScore":70.2514,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-tau2bench-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aascicode-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.6,"normalizedScore":72.0067,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-lcr-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.7,"normalizedScore":80.1849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-critpt-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-designarenawebsite-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1301,"normalizedScore":85.9736,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aagpqadiamond-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.7,"normalizedScore":86.6477,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aahle-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":25.4,"normalizedScore":44.4223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aaomniscienceindex-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-15.1,"normalizedScore":56.5934,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-omniscienceaccuracy-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29,"normalizedScore":44.3299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-omnisciencehallucinationrate-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.2,"normalizedScore":41.9783,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aaifbench-2026-07-27","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.2,"normalizedScore":85.6932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-claweval-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.8,"normalizedScore":73.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-tau2bench-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95,"normalizedScore":95.8628,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-gertlabs-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":36.68,"normalizedScore":23.3094,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-researchclawbench-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":15.3,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-sweverified-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":78,"normalizedScore":75.1381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aascicode-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42.5,"normalizedScore":70.1518,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-lcr-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.7,"normalizedScore":80.1849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-critpt-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aagpqadiamond-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87,"normalizedScore":89.9148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aahle-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.3,"normalizedScore":50.1992,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aaomniscienceindex-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.9,"normalizedScore":72.292,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-omniscienceaccuracy-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":40.5498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-omnisciencehallucinationrate-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.9,"normalizedScore":80.9409,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aaifbench-2026-07-27","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":68.8,"normalizedScore":79.2035,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-claweval-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":53.8,"normalizedScore":67.4581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-tau2bench-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-gertlabs-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":30.76,"normalizedScore":10.7988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aascicode-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.5,"normalizedScore":71.8381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-lcr-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61,"normalizedScore":80.5812,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-critpt-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aammmupro-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.8,"normalizedScore":79.5533,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-designarenawebsite-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1258,"normalizedScore":78.8779,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aagpqadiamond-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":80.9,"normalizedScore":81.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aahle-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":15.8,"normalizedScore":25.2988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aaomniscienceindex-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-19,"normalizedScore":53.5322,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-omniscienceaccuracy-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.1,"normalizedScore":44.5017,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-omnisciencehallucinationrate-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.9,"normalizedScore":35.1025,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aaifbench-2026-07-27","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":61.1,"normalizedScore":67.8466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aaagenticindex-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.71,"normalizedScore":46.263,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-tau2bench-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.9,"normalizedScore":42.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-gdpvalaa-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1079,"normalizedScore":60.5051,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-jobbench-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":8.53,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-vibecodebench-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":20.088,"normalizedScore":28.2918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aacodingindex-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.78,"normalizedScore":43.1802,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aascicode-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42.9,"normalizedScore":70.8263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-lcr-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.6,"normalizedScore":99.8679,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-critpt-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.7,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aammmupro-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.2,"normalizedScore":81.9588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-designarenawebsite-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1214,"normalizedScore":71.6172,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aagpqadiamond-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.4,"normalizedScore":87.642,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aahle-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":26.5,"normalizedScore":46.6135,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-8.1,"normalizedScore":62.0879,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.7,"normalizedScore":64.433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82.1,"normalizedScore":17.9735,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aaifbench-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.1,"normalizedScore":85.5457,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aamath500-2026-07-27","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aamath500","benchmarkName":"Artificial Analysis MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-tau2bench-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.3,"normalizedScore":94.1473,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-vibecodebench-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":1.2,"normalizedScore":1.6901,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aascicode-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":44.2,"normalizedScore":73.0185,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-lcr-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68,"normalizedScore":89.8283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-critpt-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.9,"normalizedScore":8.9783,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aammmupro-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":63.3,"normalizedScore":63.2302,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aagpqadiamond-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.3,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aahle-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":17.6,"normalizedScore":28.8845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aaomniscienceindex-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-28.7,"normalizedScore":45.9184,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-omniscienceaccuracy-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.3,"normalizedScore":37.9725,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-omnisciencehallucinationrate-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.4,"normalizedScore":29.6743,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aaifbench-2026-07-27","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":52.7,"normalizedScore":55.4572,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-claweval-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":45.2,"normalizedScore":55.4469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-tau2bench-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":91.2,"normalizedScore":92.0283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-sweverified-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.8,"normalizedScore":70.7182,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aascicode-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.7,"normalizedScore":60.371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-lcr-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-critpt-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aammmupro-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":69.9,"normalizedScore":74.5704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aagpqadiamond-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":82.8,"normalizedScore":83.9489,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aahle-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":19.9,"normalizedScore":33.4661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aaomniscienceindex-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-17.4,"normalizedScore":54.7881,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-omniscienceaccuracy-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.7,"normalizedScore":26.6323,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-omnisciencehallucinationrate-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.4,"normalizedScore":63.4499,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aaifbench-2026-07-27","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":53.5,"normalizedScore":56.6372,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-thinking-vibecodebench-2026-07-27","modelSlug":"deepseek-v3-2-thinking","modelName":"DeepSeek V3.2 (Thinking)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":5.108,"normalizedScore":7.1941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-thinking-designarenawebsite-2026-07-27","modelSlug":"deepseek-v3-2-thinking","modelName":"DeepSeek V3.2 (Thinking)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1204,"normalizedScore":69.967,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-tau2bench-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":63.7,"normalizedScore":64.2785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-gertlabs-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":47.32,"normalizedScore":45.7946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aascicode-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.6,"normalizedScore":48.398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-lcr-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22,"normalizedScore":29.0621,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-critpt-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aammmupro-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":48.4,"normalizedScore":37.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aagpqadiamond-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":63.7,"normalizedScore":56.8182,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aahle-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5,"normalizedScore":3.7849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aaomniscienceindex-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-50.9,"normalizedScore":28.4929,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-omniscienceaccuracy-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17,"normalizedScore":23.7113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-omnisciencehallucinationrate-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.8,"normalizedScore":18.3353,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aaifbench-2026-07-27","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.5,"normalizedScore":31.5634,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-tau2bench-2026-07-27","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":34.8,"normalizedScore":35.116,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aascicode-2026-07-27","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.7,"normalizedScore":60.371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-lcr-2026-07-27","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45,"normalizedScore":59.4452,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-critpt-2026-07-27","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-designarenawebsite-2026-07-27","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1152,"normalizedScore":61.3861,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aagpqadiamond-2026-07-27","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":73.5,"normalizedScore":70.7386,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aahle-2026-07-27","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.3,"normalizedScore":6.3745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aaomniscienceindex-2026-07-27","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-41.1,"normalizedScore":36.1852,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-omniscienceaccuracy-2026-07-27","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.1,"normalizedScore":34.1924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-omnisciencehallucinationrate-2026-07-27","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.5,"normalizedScore":16.2847,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aaifbench-2026-07-27","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":37.8,"normalizedScore":33.4808,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aaagenticindex-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.52,"normalizedScore":9.5472,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-tau2bench-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":24.6,"normalizedScore":24.8234,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-gdpvalaanormalized-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7,"normalizedScore":10.2941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-gdpvalaa-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":640,"normalizedScore":38.3333,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aacodingindex-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.07,"normalizedScore":18.1484,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aascicode-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.2,"normalizedScore":59.5278,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-lcr-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.7,"normalizedScore":45.8388,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-critpt-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aammmupro-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":55.7,"normalizedScore":50.1718,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aagpqadiamond-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68,"normalizedScore":62.9261,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aahle-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.1,"normalizedScore":1.992,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aaomniscienceindex-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-39.4,"normalizedScore":37.5196,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-omniscienceaccuracy-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.1,"normalizedScore":35.9107,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-omnisciencehallucinationrate-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.7,"normalizedScore":16.0434,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aaifbench-2026-07-27","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.2,"normalizedScore":31.1209,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-designarenawebsite-2026-07-27","modelSlug":"glm-4-5","modelName":"GLM-4.5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1200,"normalizedScore":69.3069,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-tau2bench-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":65.8,"normalizedScore":66.3976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-vibecodebench-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aascicode-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":44.2,"normalizedScore":73.0185,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-lcr-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.7,"normalizedScore":85.469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-critpt-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.9,"normalizedScore":8.9783,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aammmupro-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":61.8,"normalizedScore":60.6529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aagpqadiamond-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.7,"normalizedScore":86.6477,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aahle-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":17,"normalizedScore":27.6892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aaomniscienceindex-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-28.4,"normalizedScore":46.1538,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-omniscienceaccuracy-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.6,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-omnisciencehallucinationrate-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66,"normalizedScore":37.3945,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aaifbench-2026-07-27","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":50.5,"normalizedScore":52.2124,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-tau2bench-2026-07-27","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":36.5,"normalizedScore":36.8315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aascicode-2026-07-27","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.3,"normalizedScore":66.4418,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-lcr-2026-07-27","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.7,"normalizedScore":72.2589,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-critpt-2026-07-27","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aagpqadiamond-2026-07-27","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81.3,"normalizedScore":81.8182,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aahle-2026-07-27","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.9,"normalizedScore":23.506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aaomniscienceindex-2026-07-27","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-27.1,"normalizedScore":47.1743,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-omniscienceaccuracy-2026-07-27","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31,"normalizedScore":47.7663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-omnisciencehallucinationrate-2026-07-27","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84,"normalizedScore":15.6815,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aaifbench-2026-07-27","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.6,"normalizedScore":36.1357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-5-vibecodebench-2026-07-27","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":14.852,"normalizedScore":20.9174,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-preview-aacodingindex-2026-07-27","modelSlug":"o1-preview","modelName":"o1-preview","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.05,"normalizedScore":37.9081,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1-preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-tau2bench-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":90.1,"normalizedScore":90.9183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-gdpvalaanormalized-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.2,"normalizedScore":4.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-gdpvalaa-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":564,"normalizedScore":34.4949,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aaagenticindex-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.65,"normalizedScore":6.1466,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aascicode-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.1,"normalizedScore":59.3592,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aacodingindex-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.77,"normalizedScore":26.2049,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-lcr-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33,"normalizedScore":43.5931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-critpt-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-designarenawebsite-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1165,"normalizedScore":63.5314,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-mmlu-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":87.2,"normalizedScore":60.6838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-mmluproarcee-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-gpqadiamond-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":63.3,"normalizedScore":54.0591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aahle-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.7,"normalizedScore":23.1076,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aaomniscienceindex-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-44.2,"normalizedScore":33.752,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-omniscienceaccuracy-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.8,"normalizedScore":33.677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-omnisciencehallucinationrate-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.6,"normalizedScore":12.5452,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aaifbench-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":56.3,"normalizedScore":60.767,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aime2025arcee-2026-07-27","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":24,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-tau2bench-2026-07-27","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":46.5,"normalizedScore":46.9223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aascicode-2026-07-27","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":30.6,"normalizedScore":50.0843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-lcr-2026-07-27","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.7,"normalizedScore":57.7279,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-critpt-2026-07-27","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-designarenawebsite-2026-07-27","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1176,"normalizedScore":65.3465,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aagpqadiamond-2026-07-27","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":73.3,"normalizedScore":70.4545,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aahle-2026-07-27","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.8,"normalizedScore":7.3705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aaomniscienceindex-2026-07-27","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-62.5,"normalizedScore":19.3878,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-omniscienceaccuracy-2026-07-27","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.5,"normalizedScore":21.134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-omnisciencehallucinationrate-2026-07-27","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":92.3,"normalizedScore":5.6695,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aaifbench-2026-07-27","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":37.6,"normalizedScore":33.1858,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-tau2bench-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":90.1,"normalizedScore":90.9183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gertlabs-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":32.55,"normalizedScore":14.5816,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gdpvalaanormalized-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.2,"normalizedScore":4.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gdpvalaa-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":564,"normalizedScore":34.4949,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aaagenticindex-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.65,"normalizedScore":6.1466,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-sweverifiedarcee-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":63.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aascicode-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.1,"normalizedScore":59.3592,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aacodingindex-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.77,"normalizedScore":26.2049,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-lcr-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33,"normalizedScore":43.5931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-critpt-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-designarenawebsite-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1165,"normalizedScore":63.5314,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gpqadiamond-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":76.3,"normalizedScore":72.6066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-mmluproarcee-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":83.4,"normalizedScore":58.9928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aahle-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.7,"normalizedScore":23.1076,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aaomniscienceindex-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-44.2,"normalizedScore":33.752,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-omniscienceaccuracy-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.8,"normalizedScore":33.677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-omnisciencehallucinationrate-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.6,"normalizedScore":12.5452,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aaifbench-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":56.3,"normalizedScore":60.767,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aime2025arcee-2026-07-27","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":96.3,"normalizedScore":95.3826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aaagenticindex-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.27,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-tau2bench-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":10.5,"normalizedScore":10.5954,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-gdpvalaanormalized-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-gdpvalaa-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":-119,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aacodingindex-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.06,"normalizedScore":4,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aascicode-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":21.2,"normalizedScore":34.2327,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-lcr-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.7,"normalizedScore":7.5297,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-critpt-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aammmupro-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":48,"normalizedScore":36.9416,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aagpqadiamond-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":42.8,"normalizedScore":27.1307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aahle-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.7,"normalizedScore":3.1873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aaomniscienceindex-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-65.9,"normalizedScore":16.719,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-omniscienceaccuracy-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":12.5,"normalizedScore":15.9794,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.5,"normalizedScore":9.047,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aaifbench-2026-07-27","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":31.8,"normalizedScore":24.6313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aaagenticindex-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.7,"normalizedScore":8.056,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-tau2bench-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":41.2,"normalizedScore":41.5742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-gdpvalaanormalized-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.6,"normalizedScore":6.7647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-gdpvalaa-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":592,"normalizedScore":35.9091,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aacodingindex-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.64,"normalizedScore":27.4346,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aascicode-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38,"normalizedScore":62.5632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-lcr-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.7,"normalizedScore":59.0489,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-critpt-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aammmupro-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":56.8,"normalizedScore":52.0619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aagpqadiamond-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.9,"normalizedScore":75.5682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aahle-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":9.5,"normalizedScore":12.749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aaomniscienceindex-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-29.9,"normalizedScore":44.9765,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-omniscienceaccuracy-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.1,"normalizedScore":32.4742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-omnisciencehallucinationrate-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.8,"normalizedScore":36.4294,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aaifbench-2026-07-27","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":48.2,"normalizedScore":48.8201,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaagenticindex-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.17,"normalizedScore":23.4588,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-apexagentsaa-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":3.1,"normalizedScore":5.1724,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-tau2bench-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":65.8,"normalizedScore":66.3976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.1,"normalizedScore":22.2059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-gdpvalaa-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":802,"normalizedScore":46.5152,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-gertlabs-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":29.61,"normalizedScore":8.3686,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaitbench-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-reactnativeevals-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71.6,"normalizedScore":2.3904,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aacodingindex-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.44,"normalizedScore":32.8057,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aascicode-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.9,"normalizedScore":64.0809,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aalivecodebench-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aalivecodebench","benchmarkName":"Artificial Analysis LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-lcr-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.7,"normalizedScore":66.9749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-critpt-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-designarenawebsite-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":998,"normalizedScore":35.9736,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aagpqadiamond-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":78.2,"normalizedScore":77.4148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aahle-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":18.5,"normalizedScore":30.6773,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaomniscienceindex-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-50,"normalizedScore":29.1994,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.5,"normalizedScore":31.4433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.2,"normalizedScore":6.9964,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaopennessindex-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aammlupro-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaglobalmmlulite-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaifbench-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":69,"normalizedScore":79.4985,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaaime2025-2026-07-27","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaaime2025","benchmarkName":"Artificial Analysis AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-tau2bench-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83,"normalizedScore":83.7538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-vibecodebench-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":22.168,"normalizedScore":31.2212,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aascicode-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.2,"normalizedScore":66.2732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-lcr-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.3,"normalizedScore":88.9036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-critpt-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.7,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aammmupro-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.5,"normalizedScore":79.0378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aagpqadiamond-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86,"normalizedScore":88.4943,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aahle-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.4,"normalizedScore":40.4382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-6,"normalizedScore":63.7363,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.2,"normalizedScore":61.8557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.4,"normalizedScore":27.2618,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aaifbench-2026-07-27","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70,"normalizedScore":80.9735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-tau2bench-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":92.1,"normalizedScore":92.9364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-gertlabs-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":51.79,"normalizedScore":55.2409,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-jobbench-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":26,"normalizedScore":37.8384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-vibecodebench-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":37.912,"normalizedScore":53.3949,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aascicode-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":54.6,"normalizedScore":90.5565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-lcr-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-critpt-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":8.7,"normalizedScore":26.935,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aammmupro-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":76.3,"normalizedScore":85.567,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aagpqadiamond-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.9,"normalizedScore":94.0341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aahle-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":33.5,"normalizedScore":60.5578,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-2.5,"normalizedScore":66.4835,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.7,"normalizedScore":64.433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.8,"normalizedScore":29.1918,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aaifbench-2026-07-27","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":77.6,"normalizedScore":92.1829,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-tau2bench-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86.5,"normalizedScore":87.2856,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aascicode-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":41.1,"normalizedScore":67.7909,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-lcr-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.8,"normalizedScore":96.1691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-critpt-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aammmupro-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.3,"normalizedScore":82.1306,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-designarenawebsite-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1214,"normalizedScore":71.6172,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aagpqadiamond-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.2,"normalizedScore":85.9375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aahle-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.5,"normalizedScore":40.6375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.1,"normalizedScore":60.5181,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":61.3402,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.1,"normalizedScore":20.386,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aaifbench-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.6,"normalizedScore":81.8584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aamath500-2026-07-27","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aamath500","benchmarkName":"Artificial Analysis MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aacodingindex-2026-07-27","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.53,"normalizedScore":17.3852,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aascicode-2026-07-27","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":23.3,"normalizedScore":37.774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aagpqadiamond-2026-07-27","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":48.9,"normalizedScore":35.7955,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aahle-2026-07-27","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-tau2bench-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":40.9,"normalizedScore":41.2714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-gdpvalaanormalized-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-gdpvalaa-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":492,"normalizedScore":30.8586,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aaagenticindex-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.99,"normalizedScore":3.1278,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aascicode-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.6,"normalizedScore":48.398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aacodingindex-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.37,"normalizedScore":10.0919,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-lcr-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.7,"normalizedScore":44.5178,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-critpt-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aagpqadiamond-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":75.7,"normalizedScore":73.8636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aahle-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":10.2,"normalizedScore":14.1434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aaomniscienceindex-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-51.6,"normalizedScore":27.9435,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-omniscienceaccuracy-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.1,"normalizedScore":23.8832,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-omnisciencehallucinationrate-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82.9,"normalizedScore":17.0084,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aaifbench-2026-07-27","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":71.1,"normalizedScore":82.5959,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aaagenticindex-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.1,"normalizedScore":5.1464,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-apexagentsaa-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":0.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-tau2bench-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":60.2,"normalizedScore":60.7467,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.4,"normalizedScore":5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-gdpvalaa-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":567,"normalizedScore":34.6465,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-reactnativeevals-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aacodingindex-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.7,"normalizedScore":19.0389,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aascicode-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":34.4,"normalizedScore":56.4924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-lcr-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.7,"normalizedScore":40.5548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-critpt-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-designarenawebsite-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":882,"normalizedScore":16.8317,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aagpqadiamond-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68.8,"normalizedScore":64.0625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aahle-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":9.8,"normalizedScore":13.3466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aaomniscienceindex-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-63.9,"normalizedScore":18.2889,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.5,"normalizedScore":21.134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94.1,"normalizedScore":3.4982,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aaifbench-2026-07-27","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":65.1,"normalizedScore":73.7463,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aaagenticindex-2026-07-27","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.96,"normalizedScore":1.2548,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-gdpvalaanormalized-2026-07-27","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-gdpvalaa-2026-07-27","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":241,"normalizedScore":18.1818,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aascicode-2026-07-27","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":22.9,"normalizedScore":37.0995,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aacodingindex-2026-07-27","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":11.38,"normalizedScore":5.8657,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aammmupro-2026-07-27","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":41.5,"normalizedScore":25.7732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aagpqadiamond-2026-07-27","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":42.6,"normalizedScore":26.8466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aahle-2026-07-27","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4,"normalizedScore":1.7928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aaifbench-2026-07-27","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":31,"normalizedScore":23.4513,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-tau2bench-2026-07-27","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":30.7,"normalizedScore":30.9788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aascicode-2026-07-27","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.2,"normalizedScore":47.7234,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-lcr-2026-07-27","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.3,"normalizedScore":7.0013,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-critpt-2026-07-27","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aagpqadiamond-2026-07-27","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":48.6,"normalizedScore":35.3693,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aahle-2026-07-27","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4,"normalizedScore":1.7928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aaomniscienceindex-2026-07-27","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-34,"normalizedScore":41.7582,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-omniscienceaccuracy-2026-07-27","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.1,"normalizedScore":29.0378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-omnisciencehallucinationrate-2026-07-27","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.8,"normalizedScore":35.2232,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aaifbench-2026-07-27","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":31.2,"normalizedScore":23.7463,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-tau2bench-2026-07-27","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":19,"normalizedScore":19.1726,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aascicode-2026-07-27","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.9,"normalizedScore":48.9039,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-lcr-2026-07-27","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.3,"normalizedScore":32.1004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-critpt-2026-07-27","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aagpqadiamond-2026-07-27","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":51.5,"normalizedScore":39.4886,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aahle-2026-07-27","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.2,"normalizedScore":2.1912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aaomniscienceindex-2026-07-27","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-17.3,"normalizedScore":54.8666,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-omniscienceaccuracy-2026-07-27","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.3,"normalizedScore":32.8179,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-omnisciencehallucinationrate-2026-07-27","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51,"normalizedScore":55.4885,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aaifbench-2026-07-27","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39,"normalizedScore":35.2507,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen2-5-coder-32b-instruct-aascicode-2026-07-27","modelSlug":"qwen2-5-coder-32b-instruct","modelName":"Qwen2.5 Coder 32B Instruct","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":27.1,"normalizedScore":44.1821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen2-5-coder-32b-instruct-aagpqadiamond-2026-07-27","modelSlug":"qwen2-5-coder-32b-instruct","modelName":"Qwen2.5 Coder 32B Instruct","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":41.7,"normalizedScore":25.5682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen2-5-coder-32b-instruct-aahle-2026-07-27","modelSlug":"qwen2-5-coder-32b-instruct","modelName":"Qwen2.5 Coder 32B Instruct","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.8,"normalizedScore":1.3944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-super-100b-claweval-2026-07-27","modelSlug":"nemotron-3-super-100b","modelName":"Nemotron 3 Super 100B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":5.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Super 100B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aacodingindex-2026-07-27","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.63,"normalizedScore":23.1802,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aascicode-2026-07-27","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.5,"normalizedScore":48.2293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aammmupro-2026-07-27","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":55,"normalizedScore":48.9691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aagpqadiamond-2026-07-27","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":58.9,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aahle-2026-07-27","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.9,"normalizedScore":3.5857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-tau2bench-2026-07-27","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aascicode-2026-07-27","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":26,"normalizedScore":42.3272,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-lcr-2026-07-27","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-critpt-2026-07-27","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aagpqadiamond-2026-07-27","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":57.5,"normalizedScore":48.0114,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aahle-2026-07-27","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.1,"normalizedScore":1.992,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aaomniscienceindex-2026-07-27","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-56.7,"normalizedScore":23.9403,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-omniscienceaccuracy-2026-07-27","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.2,"normalizedScore":17.1821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-omnisciencehallucinationrate-2026-07-27","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.5,"normalizedScore":19.9035,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aaifbench-2026-07-27","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":23.5,"normalizedScore":12.3894,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-mini-vibecodebench-2026-07-27","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":14.171,"normalizedScore":19.9583,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-pro-aagpqadiamond-2026-07-27","modelSlug":"o3-pro","modelName":"o3-pro","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.5,"normalizedScore":86.3636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-tau2bench-2026-07-27","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":75.7,"normalizedScore":76.3875,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-sweverified-2026-07-27","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":70.8,"normalizedScore":65.1934,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aascicode-2026-07-27","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.2,"normalizedScore":59.5278,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-lcr-2026-07-27","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.3,"normalizedScore":63.8045,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-critpt-2026-07-27","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aagpqadiamond-2026-07-27","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":72.7,"normalizedScore":69.6023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aahle-2026-07-27","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7.5,"normalizedScore":8.7649,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aaomniscienceindex-2026-07-27","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-36,"normalizedScore":40.1884,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-omniscienceaccuracy-2026-07-27","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.8,"normalizedScore":35.3952,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-omnisciencehallucinationrate-2026-07-27","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":78.5,"normalizedScore":22.316,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aaifbench-2026-07-27","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":41.4,"normalizedScore":38.7906,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-turbo-aacodingindex-2026-07-27","modelSlug":"gpt-4-turbo","modelName":"GPT-4 Turbo","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.49,"normalizedScore":20.1555,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-turbo-aascicode-2026-07-27","modelSlug":"gpt-4-turbo","modelName":"GPT-4 Turbo","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":31.9,"normalizedScore":52.2766,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-turbo-aahle-2026-07-27","modelSlug":"gpt-4-turbo","modelName":"GPT-4 Turbo","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.3,"normalizedScore":0.3984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-tau2bench-2026-07-27","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":11.4,"normalizedScore":11.5035,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aascicode-2026-07-27","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":34.7,"normalizedScore":56.9983,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-lcr-2026-07-27","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.3,"normalizedScore":9.6433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-critpt-2026-07-27","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aagpqadiamond-2026-07-27","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":72.8,"normalizedScore":69.7443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aahle-2026-07-27","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":8.1,"normalizedScore":9.9602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aaomniscienceindex-2026-07-27","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-45.5,"normalizedScore":32.7316,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-omniscienceaccuracy-2026-07-27","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.9,"normalizedScore":28.6942,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-omnisciencehallucinationrate-2026-07-27","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.7,"normalizedScore":18.456,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aaifbench-2026-07-27","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":38.2,"normalizedScore":34.0708,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-0-pro-aascicode-2026-07-27","modelSlug":"gemini-1-0-pro","modelName":"Gemini 1.0 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":11.7,"normalizedScore":18.2125,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-0-pro-aagpqadiamond-2026-07-27","modelSlug":"gemini-1-0-pro","modelName":"Gemini 1.0 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":27.7,"normalizedScore":5.6818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-0-pro-aahle-2026-07-27","modelSlug":"gemini-1-0-pro","modelName":"Gemini 1.0 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-tau2bench-2026-07-27","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":21.1,"normalizedScore":21.2916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aascicode-2026-07-27","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":18.6,"normalizedScore":29.8482,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-lcr-2026-07-27","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21,"normalizedScore":27.7411,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-critpt-2026-07-27","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aammmupro-2026-07-27","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":30.8,"normalizedScore":7.3883,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aagpqadiamond-2026-07-27","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":37.4,"normalizedScore":19.4602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aahle-2026-07-27","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.9,"normalizedScore":1.5936,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aaomniscienceindex-2026-07-27","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-47.6,"normalizedScore":31.0832,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-omniscienceaccuracy-2026-07-27","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.2,"normalizedScore":24.055,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-omnisciencehallucinationrate-2026-07-27","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":78.2,"normalizedScore":22.6779,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aaifbench-2026-07-27","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.1,"normalizedScore":30.9735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-tau2bench-2026-07-27","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":71.4,"normalizedScore":72.0484,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aascicode-2026-07-27","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.9,"normalizedScore":67.4536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-lcr-2026-07-27","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.3,"normalizedScore":87.5826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-critpt-2026-07-27","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aammmupro-2026-07-27","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":67.9,"normalizedScore":71.134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aagpqadiamond-2026-07-27","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":80.9,"normalizedScore":81.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aahle-2026-07-27","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":11.9,"normalizedScore":17.5299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aaifbench-2026-07-27","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":55.4,"normalizedScore":59.4395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-tau2bench-2026-07-27","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":14,"normalizedScore":14.1271,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aascicode-2026-07-27","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":20.8,"normalizedScore":33.5582,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-lcr-2026-07-27","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19,"normalizedScore":25.0991,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-critpt-2026-07-27","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aammmupro-2026-07-27","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":44.3,"normalizedScore":30.5842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aagpqadiamond-2026-07-27","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":49.9,"normalizedScore":37.2159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aahle-2026-07-27","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.4,"normalizedScore":0.5976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aaomniscienceindex-2026-07-27","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-47.6,"normalizedScore":31.0832,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-omniscienceaccuracy-2026-07-27","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17,"normalizedScore":23.7113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-omnisciencehallucinationrate-2026-07-27","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.9,"normalizedScore":23.0398,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aaifbench-2026-07-27","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":38.1,"normalizedScore":33.9233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-tau2bench-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":8.5,"normalizedScore":8.5772,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aascicode-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":3,"normalizedScore":3.5413,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-lcr-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-critpt-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractjsonvalidity-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractjsonvalidity","benchmarkName":"Liquid image-to-JSON extraction JSON validity","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":99.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractschemaf1-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractschemaf1","benchmarkName":"Liquid image-to-JSON extraction schema consistency F1","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":99.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractvlmjudge-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractvlmjudge","benchmarkName":"Liquid image-to-JSON extraction VLM judge score","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":90.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aammmupro-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":26.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aagpqadiamond-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":28.9,"normalizedScore":7.3864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aahle-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.1,"normalizedScore":3.9841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aaomniscienceindex-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-83.9,"normalizedScore":2.5903,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-omniscienceaccuracy-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.2,"normalizedScore":3.4364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-omnisciencehallucinationrate-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94,"normalizedScore":3.6188,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aaifbench-2026-07-27","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":33.1,"normalizedScore":26.5487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-extract-liquidextractjsonvalidity-2026-07-27","modelSlug":"lfm2-5-vl-450m-extract","modelName":"LFM2.5-VL-450M-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractjsonvalidity","benchmarkName":"Liquid image-to-JSON extraction JSON validity","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":98.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-extract-liquidextractschemaf1-2026-07-27","modelSlug":"lfm2-5-vl-450m-extract","modelName":"LFM2.5-VL-450M-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractschemaf1","benchmarkName":"Liquid image-to-JSON extraction schema consistency F1","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":98.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-extract-liquidextractvlmjudge-2026-07-27","modelSlug":"lfm2-5-vl-450m-extract","modelName":"LFM2.5-VL-450M-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractvlmjudge","benchmarkName":"Liquid image-to-JSON extraction VLM judge score","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":84.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-cyber-cybergym-2026-07-27","modelSlug":"sakana-fugu-cyber","modelName":"Fugu Cyber","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":86.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-cyber-ctirealm-2026-07-27","modelSlug":"sakana-fugu-cyber","modelName":"Fugu Cyber","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-ctirealm","benchmarkName":"CTI-REALM","benchmarkCategory":"agents","benchmarkOrganisation":"Sakana AI","benchmarkVersion":"2026","score":72.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-kimiclaw247-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-kimiclaw247","benchmarkName":"Kimi Claw 24/7 Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":46.9,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-mcpatlas-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":76,"normalizedScore":79.3515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-mcpmarkverified-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mcpmarkverified","benchmarkName":"MCPMark-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"MCPMark","benchmarkVersion":"2026","score":81.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aaagenticindex-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.59,"normalizedScore":53.3188,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-tau2bench-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":90.1,"normalizedScore":90.9183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-gdpvalaanormalized-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.3,"normalizedScore":50.4412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-gdpvalaa-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1186,"normalizedScore":65.9091,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-kimicodebenchv2-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"kimi-code-bench-v2","benchmarkName":"Kimi Code Bench v2","benchmarkCategory":"coding","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":62,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-programbench-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-programbench","benchmarkName":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","benchmarkCategory":"coding","benchmarkOrganisation":"John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","benchmarkVersion":"2026","score":53.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-mlsbenchlite-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mlsbenchlite","benchmarkName":"MLS-Bench Lite","benchmarkCategory":"coding","benchmarkOrganisation":"MLS-Bench","benchmarkVersion":"2026","score":35.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aacodingindex-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.76,"normalizedScore":75.6608,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aascicode-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.5,"normalizedScore":78.5835,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-lcr-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.3,"normalizedScore":87.5826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-critpt-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":10,"normalizedScore":30.9598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-designarenawebsite-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1302,"normalizedScore":86.1386,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aagpqadiamond-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.6,"normalizedScore":93.608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aahle-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":32.8,"normalizedScore":59.1633,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aaomniscienceindex-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.7,"normalizedScore":60.0471,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-omniscienceaccuracy-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.6,"normalizedScore":60.8247,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-omnisciencehallucinationrate-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.3,"normalizedScore":20.1448,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aaifbench-2026-07-27","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":63.1,"normalizedScore":70.7965,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-tau2bench-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83,"normalizedScore":83.7538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-gertlabs-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":49.68,"normalizedScore":50.7819,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-jobbench-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":26.2,"normalizedScore":38.2716,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-vibecodebench-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":13.115,"normalizedScore":18.4711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aascicode-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.2,"normalizedScore":66.2732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-lcr-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.3,"normalizedScore":88.9036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-critpt-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.7,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aammmupro-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.5,"normalizedScore":79.0378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-designarenawebsite-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1191,"normalizedScore":67.8218,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aagpqadiamond-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86,"normalizedScore":88.4943,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aahle-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.4,"normalizedScore":40.4382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aaomniscienceindex-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-6,"normalizedScore":63.7363,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-omniscienceaccuracy-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.2,"normalizedScore":61.8557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-omnisciencehallucinationrate-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.4,"normalizedScore":27.2618,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aaifbench-2026-07-27","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70,"normalizedScore":80.9735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-cyber-cybergym-2026-07-27","modelSlug":"gemini-3-5-flash-cyber","modelName":"Gemini 3.5 Flash Cyber","providerId":"google","providerName":"Google","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":83.2,"normalizedScore":91.5332,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash Cyber; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-1-35b-a3b-androidworld-2026-07-27","modelSlug":"holo3-1-35b-a3b","modelName":"Holo3.1-35B-A3B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":79.3,"normalizedScore":84.1121,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3.1-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-1-4b-androidworld-2026-07-27","modelSlug":"holo3-1-4b","modelName":"Holo3.1-4B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":71,"normalizedScore":6.5421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3.1-4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-1-9b-androidworld-2026-07-27","modelSlug":"holo3-1-9b","modelName":"Holo3.1-9B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":71,"normalizedScore":6.5421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3.1-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-tau2bench-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74.3,"normalizedScore":74.9748,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-gertlabs-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":43.74,"normalizedScore":38.2291,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-vibecodebench-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":3.506,"normalizedScore":4.9378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aascicode-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.3,"normalizedScore":63.0691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-lcr-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.7,"normalizedScore":61.6909,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-critpt-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-designarenawebsite-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1148,"normalizedScore":60.7261,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aagpqadiamond-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.4,"normalizedScore":74.858,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aahle-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":11.1,"normalizedScore":15.9363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aaomniscienceindex-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-43.1,"normalizedScore":34.6154,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-omniscienceaccuracy-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.4,"normalizedScore":36.4261,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-omnisciencehallucinationrate-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.4,"normalizedScore":9.1677,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aaifbench-2026-07-27","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":44.1,"normalizedScore":42.7729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-fast-reactnativeevals-2026-07-27","modelSlug":"composer-2-fast","modelName":"Composer 2 Fast","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":94.9,"normalizedScore":95.2191,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-tau2bench-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":89.5,"normalizedScore":90.3128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-vibecodebench-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":20.63,"normalizedScore":29.0551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aascicode-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49.5,"normalizedScore":81.9562,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-lcr-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-critpt-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.6,"normalizedScore":14.2415,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aammmupro-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74,"normalizedScore":81.6151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-designarenawebsite-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1277,"normalizedScore":82.0132,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aagpqadiamond-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86.6,"normalizedScore":89.3466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aahle-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.4,"normalizedScore":50.3984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aaomniscienceindex-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.3,"normalizedScore":78.8854,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-omniscienceaccuracy-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.7,"normalizedScore":73.0241,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-omnisciencehallucinationrate-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59.8,"normalizedScore":44.8733,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aammlupro-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.5,"normalizedScore":96.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aaglobalmmlulite-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.3,"normalizedScore":81.7308,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aaifbench-2026-07-27","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":58,"normalizedScore":63.2743,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-thinking-vibecodebench-2026-07-27","modelSlug":"claude-haiku-4-5-thinking","modelName":"Claude Haiku 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":11.393,"normalizedScore":16.0458,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-thinking-designarenawebsite-2026-07-27","modelSlug":"claude-haiku-4-5-thinking","modelName":"Claude Haiku 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1152,"normalizedScore":61.3861,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-thinking-vibecodebench-2026-07-27","modelSlug":"claude-sonnet-4-5-thinking","modelName":"Claude Sonnet 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":22.621,"normalizedScore":31.8592,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-thinking-designarenawebsite-2026-07-27","modelSlug":"claude-sonnet-4-5-thinking","modelName":"Claude Sonnet 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1219,"normalizedScore":72.4422,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-235b-a22b-screenspotpro-2026-07-27","modelSlug":"holo2-235b-a22b","modelName":"Holo2-235B-A22B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":70.6,"normalizedScore":59.0047,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-235B-A22B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-30b-a3b-screenspotpro-2026-07-27","modelSlug":"holo2-30b-a3b","modelName":"Holo2-30B-A3B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":66.1,"normalizedScore":48.3412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-4b-screenspotpro-2026-07-27","modelSlug":"holo2-4b","modelName":"Holo2-4B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":57.2,"normalizedScore":27.2512,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-8b-screenspotpro-2026-07-27","modelSlug":"holo2-8b","modelName":"Holo2-8B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":58.9,"normalizedScore":31.2796,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aaagenticindex-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.73,"normalizedScore":55.3919,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-gdpvalaanormalized-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.8,"normalizedScore":52.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-gdpvalaa-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1215,"normalizedScore":67.3737,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aascicode-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.6,"normalizedScore":78.7521,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aacodingindex-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.8,"normalizedScore":72.8905,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-lcr-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-critpt-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.9,"normalizedScore":15.1703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-designarenawebsite-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1229,"normalizedScore":74.0924,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aagpqadiamond-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.7,"normalizedScore":93.75,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aahle-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":31.6,"normalizedScore":56.7729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aaomniscienceindex-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-18.5,"normalizedScore":53.9246,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-omniscienceaccuracy-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.5,"normalizedScore":48.6254,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-omnisciencehallucinationrate-2026-07-27","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73,"normalizedScore":28.9505,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-colbert-350m-nanobeirmultilingual-2026-07-27","modelSlug":"lfm2-5-colbert-350m","modelName":"LFM2.5-ColBERT-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-nanobeirmultilingual","benchmarkName":"NanoBEIR Multilingual Extended","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":60.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-colbert-350m-mkqa11-2026-07-27","modelSlug":"lfm2-5-colbert-350m","modelName":"LFM2.5-ColBERT-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mkqa11","benchmarkName":"MKQA-11 multilingual retrieval","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":69.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-embedding-350m-nanobeirmultilingual-2026-07-27","modelSlug":"lfm2-5-embedding-350m","modelName":"LFM2.5-Embedding-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-nanobeirmultilingual","benchmarkName":"NanoBEIR Multilingual Extended","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":57.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-embedding-350m-mkqa11-2026-07-27","modelSlug":"lfm2-5-embedding-350m","modelName":"LFM2.5-Embedding-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mkqa11","benchmarkName":"MKQA-11 multilingual retrieval","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":69.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-build-0-1-gertlabs-2026-07-27","modelSlug":"grok-build-0-1","modelName":"Grok Build 0.1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":49.15,"normalizedScore":49.6619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Build 0.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-tau2bench-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":20.8,"normalizedScore":20.9889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aaagenticindex-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.51,"normalizedScore":2.255,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-gdpvalaanormalized-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-gdpvalaa-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":88,"normalizedScore":10.4545,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aascicode-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":20.9,"normalizedScore":33.7268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aacodingindex-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.23,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-lcr-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15,"normalizedScore":19.8151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-critpt-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aammmupro-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":44.6,"normalizedScore":31.0997,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-gpqa-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":43.4,"normalizedScore":25.667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-mmlupro-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":60,"normalizedScore":57.8828,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aagpqadiamond-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":40.5,"normalizedScore":23.8636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aahle-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.8,"normalizedScore":3.3865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aaomniscienceindex-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-24,"normalizedScore":49.6075,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-omniscienceaccuracy-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.7,"normalizedScore":6.0137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.9,"normalizedScore":77.3221,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aaifbench-2026-07-27","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36,"normalizedScore":30.826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-tau2bench-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":20.8,"normalizedScore":20.9889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aaagenticindex-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.79,"normalizedScore":2.7641,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-gdpvalaanormalized-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-gdpvalaa-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":231,"normalizedScore":17.6768,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aascicode-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":24.4,"normalizedScore":39.629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aacodingindex-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.39,"normalizedScore":3.053,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-lcr-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.7,"normalizedScore":40.5548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-critpt-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aammmupro-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":51.4,"normalizedScore":42.7835,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-gpqa-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":58.6,"normalizedScore":47.3534,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-mmlupro-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":69.4,"normalizedScore":71.2578,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aagpqadiamond-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":52.2,"normalizedScore":40.483,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aahle-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.7,"normalizedScore":1.1952,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aaomniscienceindex-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-20,"normalizedScore":52.7473,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-omniscienceaccuracy-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.6,"normalizedScore":9.2784,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-omnisciencehallucinationrate-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.3,"normalizedScore":79.2521,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aaifbench-2026-07-27","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":40.6,"normalizedScore":37.6106,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-tau2bench-2026-07-27","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":46.8,"normalizedScore":47.225,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aascicode-2026-07-27","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":26.4,"normalizedScore":43.0017,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-lcr-2026-07-27","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-critpt-2026-07-27","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aagpqadiamond-2026-07-27","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":73.8,"normalizedScore":71.1648,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aahle-2026-07-27","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":10.1,"normalizedScore":13.9442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aaomniscienceindex-2026-07-27","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-59.5,"normalizedScore":21.7425,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-omniscienceaccuracy-2026-07-27","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.6,"normalizedScore":24.7423,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-omnisciencehallucinationrate-2026-07-27","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.5,"normalizedScore":4.222,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aaifbench-2026-07-27","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":34.4,"normalizedScore":28.4661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-tau2bench-2026-07-27","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":34.5,"normalizedScore":34.8133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aascicode-2026-07-27","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":19.2,"normalizedScore":30.86,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-lcr-2026-07-27","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-critpt-2026-07-27","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aagpqadiamond-2026-07-27","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":63.3,"normalizedScore":56.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aahle-2026-07-27","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7,"normalizedScore":7.7689,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aaomniscienceindex-2026-07-27","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-72,"normalizedScore":11.9309,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-omniscienceaccuracy-2026-07-27","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":12.7,"normalizedScore":16.323,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-omnisciencehallucinationrate-2026-07-27","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":97,"normalizedScore":0,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aaifbench-2026-07-27","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":26.5,"normalizedScore":16.8142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-tau2bench-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":24.3,"normalizedScore":24.5207,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aascicode-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.1,"normalizedScore":54.3002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-lcr-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28,"normalizedScore":36.9881,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-critpt-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aammmupro-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":53,"normalizedScore":45.5326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-designarenawebsite-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1109,"normalizedScore":54.2904,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aagpqadiamond-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":57.8,"normalizedScore":48.4375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aahle-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.3,"normalizedScore":2.3904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aaomniscienceindex-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-31.5,"normalizedScore":43.7206,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-omniscienceaccuracy-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.3,"normalizedScore":25.945,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-omnisciencehallucinationrate-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.9,"normalizedScore":43.5464,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aaifbench-2026-07-27","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.3,"normalizedScore":35.6932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-tau2bench-2026-07-27","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":22.8,"normalizedScore":23.0071,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aascicode-2026-07-27","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":8.7,"normalizedScore":13.1535,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-lcr-2026-07-27","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4,"normalizedScore":5.284,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-critpt-2026-07-27","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aagpqadiamond-2026-07-27","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":28.1,"normalizedScore":6.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aahle-2026-07-27","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.1,"normalizedScore":3.9841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aaomniscienceindex-2026-07-27","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-81.8,"normalizedScore":4.2386,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-omniscienceaccuracy-2026-07-27","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.1,"normalizedScore":4.9828,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-omnisciencehallucinationrate-2026-07-27","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.5,"normalizedScore":4.222,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aaifbench-2026-07-27","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":20.5,"normalizedScore":7.9646,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-tau2bench-2026-07-27","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":13.2,"normalizedScore":13.3199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aascicode-2026-07-27","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-lcr-2026-07-27","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-critpt-2026-07-27","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aagpqadiamond-2026-07-27","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":23.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aahle-2026-07-27","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.7,"normalizedScore":5.1793,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aaomniscienceindex-2026-07-27","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-72.1,"normalizedScore":11.8524,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-omniscienceaccuracy-2026-07-27","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-omnisciencehallucinationrate-2026-07-27","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.8,"normalizedScore":23.1604,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aaifbench-2026-07-27","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":15.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-tau2bench-2026-07-27","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":19.6,"normalizedScore":19.778,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aascicode-2026-07-27","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":8.2,"normalizedScore":12.3103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-lcr-2026-07-27","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.3,"normalizedScore":8.3223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-critpt-2026-07-27","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aagpqadiamond-2026-07-27","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":24.6,"normalizedScore":1.2784,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aahle-2026-07-27","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5,"normalizedScore":3.7849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aaomniscienceindex-2026-07-27","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-73.6,"normalizedScore":10.675,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-omniscienceaccuracy-2026-07-27","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.3,"normalizedScore":3.6082,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-omnisciencehallucinationrate-2026-07-27","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.4,"normalizedScore":16.4053,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aaifbench-2026-07-27","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":25.1,"normalizedScore":14.7493,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-tau2bench-2026-07-27","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":14.6,"normalizedScore":14.7326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aascicode-2026-07-27","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":1.7,"normalizedScore":1.3491,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-lcr-2026-07-27","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-critpt-2026-07-27","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aagpqadiamond-2026-07-27","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":29,"normalizedScore":7.5284,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aahle-2026-07-27","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.4,"normalizedScore":6.5737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aaomniscienceindex-2026-07-27","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-87.2,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-omniscienceaccuracy-2026-07-27","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.7,"normalizedScore":0.8591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-omnisciencehallucinationrate-2026-07-27","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94.4,"normalizedScore":3.1363,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aaifbench-2026-07-27","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":17.1,"normalizedScore":2.9499,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aascicode-2026-07-27","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.6,"normalizedScore":61.8887,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-lcr-2026-07-27","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.7,"normalizedScore":12.8137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aagpqadiamond-2026-07-27","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":61.5,"normalizedScore":53.6932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aahle-2026-07-27","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.5,"normalizedScore":4.7809,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aaifbench-2026-07-27","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":22.9,"normalizedScore":11.5044,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-pro-gpqa-2026-07-27","modelSlug":"o1-pro","modelName":"o1-pro","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":79,"normalizedScore":76.4588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1-pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-tau2bench-2026-07-27","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":20.5,"normalizedScore":20.6862,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aascicode-2026-07-27","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":7.4,"normalizedScore":10.9612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-lcr-2026-07-27","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-critpt-2026-07-27","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aagpqadiamond-2026-07-27","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":42.4,"normalizedScore":26.5625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aahle-2026-07-27","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.8,"normalizedScore":5.3785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aaomniscienceindex-2026-07-27","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-82.6,"normalizedScore":3.6107,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-omniscienceaccuracy-2026-07-27","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.7,"normalizedScore":2.5773,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-omnisciencehallucinationrate-2026-07-27","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.5,"normalizedScore":6.6345,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aaifbench-2026-07-27","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":25.3,"normalizedScore":15.0442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-tau2bench-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74.3,"normalizedScore":74.9748,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aaagenticindex-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8,"normalizedScore":14.0571,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-gdpvalaanormalized-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.9,"normalizedScore":7.2059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-gdpvalaa-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":598,"normalizedScore":36.2121,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aascicode-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.6,"normalizedScore":58.516,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aacodingindex-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.11,"normalizedScore":35.1661,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-lcr-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.7,"normalizedScore":73.5799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-critpt-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aagpqadiamond-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":78.3,"normalizedScore":77.5568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aahle-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":13.1,"normalizedScore":19.9203,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aaomniscienceindex-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-57.9,"normalizedScore":22.9984,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-omniscienceaccuracy-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.5,"normalizedScore":22.8522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-omnisciencehallucinationrate-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.1,"normalizedScore":9.5296,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aaifbench-2026-07-27","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":64.7,"normalizedScore":73.1563,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-tau2bench-2026-07-27","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":31.9,"normalizedScore":32.1897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aascicode-2026-07-27","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":24.8,"normalizedScore":40.3035,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-lcr-2026-07-27","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-critpt-2026-07-27","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aagpqadiamond-2026-07-27","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":46.0227,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aahle-2026-07-27","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.8,"normalizedScore":1.3944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aaomniscienceindex-2026-07-27","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-61.7,"normalizedScore":20.0157,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-omniscienceaccuracy-2026-07-27","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.6,"normalizedScore":21.3058,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-omnisciencehallucinationrate-2026-07-27","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.5,"normalizedScore":6.6345,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aaifbench-2026-07-27","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":33.7,"normalizedScore":27.4336,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.6.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-07-27","publishedAt":"2026-07-27","sourceId":"benchlm-public-dataset-2026-07-27","sourceTitle":"BenchLM public datasets — 2026-07-27","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-07-27","sourceCheckedAt":"2026-07-27","checkedAt":"2026-07-27","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-terminalbench2-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":88,"normalizedScore":93.0605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-osworldverified-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":85,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-browsecomp-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":88,"normalizedScore":91.2134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-exploitgym-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":17.5,"normalizedScore":50.7599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-sweverified-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":95.5,"normalizedScore":99.3094,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-swepro-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":80.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-swemultimodal-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultimodal","benchmarkName":"SWE-bench Multimodal","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":54.9,"normalizedScore":85.623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-charxiv-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":93.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-charxivnotools-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":88.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-gpqa-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":94.1,"normalizedScore":98.0026,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-hle-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":64.5,"normalizedScore":99.6491,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-hlenotools-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":59,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-swemultilingual-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":92.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-mythos-5-usamo2026-2026-08-01","modelSlug":"claude-mythos-5","modelName":"Claude Mythos 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":97.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Mythos 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-frontierbench-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontierbench","benchmarkName":"FrontierBench v0.1","benchmarkCategory":"agents","benchmarkOrganisation":"Ryan Marten, Alex Shaw, Andy Konwinski, Harbor, and the Laude Institute","benchmarkVersion":"2026","score":43.3,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-browsecomp-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":90.8,"normalizedScore":97.0711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-hlewithtools-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":64.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-deepsearchqa-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-draco-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-draco","benchmarkName":"Data Research and Analysis with Complex Operations","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":88.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-multiagentbrowsecompprerelease-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-multiagentbrowsecompprerelease","benchmarkName":"Multi-Agent BrowseComp — 10-agent team prerelease configuration","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":93.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-osworld2-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":70.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-mcpatlas-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":85.8,"normalizedScore":96.0751,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-mcpatlasclaimcoverage-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mcpatlasclaimcoverage","benchmarkName":"MCP-Atlas mean claim coverage","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":89.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-legalagentbenchallpass-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-legalagentbenchallpass","benchmarkName":"Legal Agent Benchmark all-pass rate — Anthropic harness","benchmarkCategory":"agents","benchmarkOrganisation":"Harvey AI and Anthropic","benchmarkVersion":"2026","score":23.58,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-legalagentbenchcriterionpass-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-legalagentbenchcriterionpass","benchmarkName":"Legal Agent Benchmark mean criterion-pass rate — Anthropic harness","benchmarkCategory":"agents","benchmarkOrganisation":"Harvey AI and Anthropic","benchmarkVersion":"2026","score":93.74,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-legalagentbenchheldoutallpass-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-legalagentbenchheldoutallpass","benchmarkName":"Legal Agent Benchmark all-pass rate — Harvey held-out set","benchmarkCategory":"agents","benchmarkOrganisation":"Harvey AI","benchmarkVersion":"2026","score":11.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-legalagentbenchheldoutcriterionpass-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-legalagentbenchheldoutcriterionpass","benchmarkName":"Legal Agent Benchmark mean criterion-pass rate — Harvey held-out set","benchmarkCategory":"agents","benchmarkOrganisation":"Harvey AI","benchmarkVersion":"2026","score":94.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-gdpvalaa-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1862,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-toolathlonverified-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-toolathlonverified","benchmarkName":"Toolathlon-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":80.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-toolathlonverifiedpass3-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-toolathlonverifiedpass3","benchmarkName":"Toolathlon Verified Pass@3","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":87,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-toolathlonverifiedpass3all-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-toolathlonverifiedpass3all","benchmarkName":"Toolathlon Verified Pass cubed","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":73.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-toolathlonverifiedavgturns-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-toolathlonverifiedavgturns","benchmarkName":"Toolathlon Verified average assistant turns","benchmarkCategory":"agents","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":23.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-automationbench-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-automationbench","benchmarkName":"AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":26,"normalizedScore":15.7895,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aaagenticindex-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.26,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-gdpvalaanormalized-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aatau3banking-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.3,"normalizedScore":83.6842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aabriefcaseelo-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1720,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aaterminalbench21-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-sweverified-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":96,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-swepro-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":79.2,"normalizedScore":97.0588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-swemultilingual-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":89.5,"normalizedScore":93.2836,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-swemultimodal-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultimodal","benchmarkName":"SWE-bench Multimodal","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":59.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-deepswe-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":68.8,"normalizedScore":87.9257,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-frontiercode-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":53.4,"normalizedScore":99.6575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-frontiercode11extended-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode11extended","benchmarkName":"FrontierCode 1.1 Extended","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":63.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-programbenchepisode1-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-programbenchepisode1","benchmarkName":"ProgramBench hidden-test pass rate after episode 1","benchmarkCategory":"coding","benchmarkOrganisation":"Yang et al.","benchmarkVersion":"2026","score":83,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-programbench-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-programbench","benchmarkName":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","benchmarkCategory":"coding","benchmarkOrganisation":"John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","benchmarkVersion":"2026","score":93,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aacodingindex-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.98,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aascicode-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":55.7,"normalizedScore":92.4115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-arcagi1-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-arcagi1","benchmarkName":"ARC-AGI-1 Semi-Private Evaluation","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":97.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-arcagi2-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":90.4,"normalizedScore":97.3384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-arcagi3-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":30.16,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-lcr-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70,"normalizedScore":92.4703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-critpt-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":29.1,"normalizedScore":90.0929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-chartography-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-chartography","benchmarkName":"Chartography without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":29.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-chartographywithtools-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-chartographywithtools","benchmarkName":"Chartography with image and code tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":83,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-benchcadvision2code-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-benchcadvision2code","benchmarkName":"BenchCAD Vision2Code voxel IoU without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Zhang et al. and Anthropic","benchmarkVersion":"2026","score":0.366,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-benchcadvision2codewithtools-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-benchcadvision2codewithtools","benchmarkName":"BenchCAD Vision2Code voxel IoU with tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Zhang et al. and Anthropic","benchmarkVersion":"2026","score":0.821,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-gdppdf-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdppdf","benchmarkName":"GDP.pdf mean criteria pass rate without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":83.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-gdppdfwithtools-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdppdfwithtools","benchmarkName":"GDP.pdf mean criteria pass rate with tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":85.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-officeqa-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-officeqa","benchmarkName":"OfficeQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Databricks and Anthropic","benchmarkVersion":"2026","score":78.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-officeqapro-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":66.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aammmupro-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":84.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-designarenawebsite-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1341,"normalizedScore":94.0199,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-hle-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":64.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-hlenotools-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":56.3,"normalizedScore":94.9814,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-healthbench-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-healthbench","benchmarkName":"HealthBench raw score","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":67.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-healthbenchlengthadjusted-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-healthbenchlengthadjusted","benchmarkName":"HealthBench length-adjusted score","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":57.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-healthbenchprofessional-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":59.8,"normalizedScore":94.3548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-healthbenchprofessionalraw-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-healthbenchprofessionalraw","benchmarkName":"HealthBench Professional raw score","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":73.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-biomysterybenchhumansolvable-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-biomysterybenchhumansolvable","benchmarkName":"BioMysteryBench Human Solvable","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":90.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-biomysterybenchhumandifficult-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-biomysterybenchhumandifficult","benchmarkName":"BioMysteryBench Human Difficult","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":49.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-spatialbenchverified-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-spatialbenchverified","benchmarkName":"LatchBio SpatialBench Verified","benchmarkCategory":"knowledge","benchmarkOrganisation":"LatchBio and Anthropic","benchmarkVersion":"2026","score":72.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-singlecellbench-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-singlecellbench","benchmarkName":"LatchBio SingleCellBench","benchmarkCategory":"knowledge","benchmarkOrganisation":"LatchBio and Anthropic","benchmarkVersion":"2026","score":60.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-proteingymhard-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-proteingymhard","benchmarkName":"ProteinGym Hard","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":47.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-proteindesign-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-proteindesign","benchmarkName":"Anthropic Protein Design evaluation","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":42.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-organicchemistryv2-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-organicchemistryv2","benchmarkName":"Anthropic Organic Chemistry V2 evaluation","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":61.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-protocolstroubleshooting-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-protocolstroubleshooting","benchmarkName":"Molecular Biology Protocols Troubleshooting","benchmarkCategory":"knowledge","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":61.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-protocolsunderstanding-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-protocolsunderstanding","benchmarkName":"Benchling Molecular Biology Protocols Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Benchling and Anthropic","benchmarkVersion":"2026","score":78.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aagpqadiamond-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":93.2,"normalizedScore":98.7216,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-aaomniscienceindex-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.3,"normalizedScore":93.0141,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-omniscienceaccuracy-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.2,"normalizedScore":87.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-omnisciencehallucinationrate-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.1,"normalizedScore":56.5742,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-gmmlu-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gmmlu","benchmarkName":"Global MMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"Singh et al.","benchmarkVersion":"2024","score":92.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-milu-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-milu","benchmarkName":"Multi-task Indic Language Understanding Benchmark","benchmarkCategory":"knowledge","benchmarkOrganisation":"Verma et al.","benchmarkVersion":"2024","score":92.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-include-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-include","benchmarkName":"INCLUDE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":89.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-imo2026-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-imo2026","benchmarkName":"International Mathematical Olympiad 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Anthropic","benchmarkVersion":"2026","score":42,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-riemannbench-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-riemannbench","benchmarkName":"RiemannBench without tools","benchmarkCategory":"mathematics","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":60,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-riemannbenchwithtools-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-riemannbenchwithtools","benchmarkName":"RiemannBench with tools","benchmarkCategory":"mathematics","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":79,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-arxivmathjune2026-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-arxivmathjune2026","benchmarkName":"ArXivMath June 2026 without tools","benchmarkCategory":"mathematics","benchmarkOrganisation":"MathArena and Anthropic","benchmarkVersion":"2026","score":90.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-5-arxivmathjune2026withtools-2026-08-01","modelSlug":"claude-opus-5","modelName":"Claude Opus 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-arxivmathjune2026withtools","benchmarkName":"ArXivMath June 2026 with tools","benchmarkCategory":"mathematics","benchmarkOrganisation":"MathArena and Anthropic","benchmarkVersion":"2026","score":91.3,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-terminalbench2-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":88.3,"normalizedScore":93.5943,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-browsecomp-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":91.2,"normalizedScore":97.9079,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-toolathlonverified-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-toolathlonverified","benchmarkName":"Toolathlon-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":73.2,"normalizedScore":76.0518,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mcpatlas-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":84.2,"normalizedScore":93.3447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-automationbench-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-automationbench","benchmarkName":"AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":30.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-jobbench-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":52.9,"normalizedScore":96.1014,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-apexagents-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-apexagents","benchmarkName":"APEX-Agents","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI / APEX-Agents benchmark authors","benchmarkVersion":"2026","score":37.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-deckbench-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deckbench","benchmarkName":"DECK-Bench (Internal)","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":73.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaagenticindex-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.07,"normalizedScore":90.5619,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gdpvalaanormalized-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59.4,"normalizedScore":87.2247,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gdpvalaa-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1687,"normalizedScore":91.1661,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aabriefcaseelo-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1540,"normalizedScore":85.0498,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaautomationbench-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaenterpriseopsgym-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.3,"normalizedScore":77.3437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aatau3banking-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaterminalbench21-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85,"normalizedScore":87.9412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-apexagentsaa-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":41.3,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaitbench-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.7,"normalizedScore":83.2016,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-deepswe-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":67.5,"normalizedScore":83.9009,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-frontierswe-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"frontierswe","benchmarkName":"FrontierSWE","benchmarkCategory":"coding","benchmarkOrganisation":"FrontierSWE","benchmarkVersion":"2026","score":81.2,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-kimicodebenchv2-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"kimi-code-bench-v2","benchmarkName":"Kimi Code Bench v2","benchmarkCategory":"coding","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":72.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-swemarathon-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-marathon","benchmarkName":"SWE Marathon","benchmarkCategory":"coding","benchmarkOrganisation":"Abundant AI","benchmarkVersion":"2026","score":42,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-posttrainbench-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"posttrain-bench","benchmarkName":"PostTrainBench","benchmarkCategory":"coding","benchmarkOrganisation":"PostTrainBench","benchmarkVersion":"2026","score":36.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aacodingindex-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76.24,"normalizedScore":97.5406,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-vulcanbench-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-vulcanbench","benchmarkName":"VulcanBench v3","benchmarkCategory":"coding","benchmarkOrganisation":"VulcanBench contributors","benchmarkVersion":"2026","score":73.91,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-lcr-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.7,"normalizedScore":98.679,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-critpt-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":23.4,"normalizedScore":72.4458,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-officeqapro-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":63.3,"normalizedScore":84.5494,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mmmupro-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81.6,"normalizedScore":60,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-mmmupropython-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.4,"normalizedScore":92.053,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-charxivnotools-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":84.8,"normalizedScore":65.5462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-babyvisionpython-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-babyvisionpython","benchmarkName":"BabyVision with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":85.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aammmupro-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.5,"normalizedScore":92.7835,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-designarenawebsite-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1377,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gpqa-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":93.5,"normalizedScore":97.1465,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-gpqadiamond-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":93.5,"normalizedScore":97.1465,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-hlenotools-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":43.5,"normalizedScore":71.1896,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-aaomniscienceindex-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.4,"normalizedScore":82.8885,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-omniscienceaccuracy-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46,"normalizedScore":73.5395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-omnisciencehallucinationrate-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.9,"normalizedScore":55.6092,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-exploitbench-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-exploitbench","benchmarkName":"ExploitBench v8-bench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Seunghyun Lee, David Brumley, Carnegie Mellon University","benchmarkVersion":"2026","score":32,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-acecyberrangesolved-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-acecyberrangesolved","benchmarkName":"ACE Cyber Range Challenges Solved","benchmarkCategory":"knowledge","benchmarkOrganisation":"NIST CAISI and UK AISI","benchmarkVersion":"2026","score":0,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-lastonescyberrangesteps-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-lastonescyberrangesteps","benchmarkName":"The Last Ones Average Progress","benchmarkCategory":"knowledge","benchmarkOrganisation":"NIST CAISI and UK AISI","benchmarkVersion":"2026","score":17,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-3-lastonescyberrangecompletion-2026-08-01","modelSlug":"kimi-k3","modelName":"Kimi K3","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-lastonescyberrangecompletion","benchmarkName":"The Last Ones Cyber Range Completion Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"NIST CAISI and UK AISI","benchmarkVersion":"2026","score":10,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-terminalbench2-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":91.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-browsecomp-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":92.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-osworld2-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":62.6,"normalizedScore":88.2006,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-cybergym-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":84.5,"normalizedScore":94.508,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-exploitgym-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":33.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-toolathlon-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":58,"normalizedScore":63.8604,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaagenticindex-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54,"normalizedScore":97.7087,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-tau2bench-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":85.1,"normalizedScore":85.8729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61.8,"normalizedScore":90.7489,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gdpvalaa-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1735,"normalizedScore":93.5891,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aabriefcaseelo-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1503,"normalizedScore":81.9767,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaitbench-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aatau3banking-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33,"normalizedScore":97.8947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaautomationbench-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.2,"normalizedScore":92.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaharveylab-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.2,"normalizedScore":90.8302,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-terminalbenchhard-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":65.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaterminalbench21-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88,"normalizedScore":96.7647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaenterpriseopsgym-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.9,"normalizedScore":67.9687,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-swepro-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":64.6,"normalizedScore":58.0214,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-deepswe-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":72.7,"normalizedScore":100,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiercode11extended-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode11extended","benchmarkName":"FrontierCode 1.1 Extended","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":60.6,"normalizedScore":64.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-vulcanbench-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vulcanbench","benchmarkName":"VulcanBench v3","benchmarkCategory":"coding","benchmarkOrganisation":"VulcanBench contributors","benchmarkVersion":"2026","score":87,"normalizedScore":75.2731,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aacodingindex-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.39,"normalizedScore":99.1661,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aascicode-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":93.086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-arcagi2-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":92.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-arcagi3-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":7.78,"normalizedScore":25.5737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-genebenchpro-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-genebenchpro","benchmarkName":"GeneBench-Pro","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":28.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-lcr-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.7,"normalizedScore":97.358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-critpt-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":32.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-mmmupro-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":83,"normalizedScore":64.5161,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-mmmupropython-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":84.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aammmupro-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":83.4,"normalizedScore":97.7663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gpqa-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":94.6,"normalizedScore":98.7159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-gpqadiamond-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":94.6,"normalizedScore":98.7159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-healthbenchprofessional-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":60.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-healthbenchhard-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":33.1,"normalizedScore":65.3571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aahle-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":47.2,"normalizedScore":87.8486,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.7,"normalizedScore":85.4788,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.5,"normalizedScore":95.0172,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.8,"normalizedScore":9.8914,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-aaifbench-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.7,"normalizedScore":84.9558,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiermath-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":89,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":89,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-frontiermathv2tier4-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":83,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-sol-exploitbench-2026-08-01","modelSlug":"gpt-5-6-sol","modelName":"GPT-5.6 Sol","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-exploitbench","benchmarkName":"ExploitBench v8-bench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Seunghyun Lee, David Brumley, Carnegie Mellon University","benchmarkVersion":"2026","score":73.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Sol; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-terminalbench2-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":84.3,"normalizedScore":86.4769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-osworldverified-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":85,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-gdpvalaa-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1747,"normalizedScore":94.1949,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaagenticindex-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.81,"normalizedScore":95.5446,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-tau2bench-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-gdpvalaanormalized-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.3,"normalizedScore":91.4831,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aabriefcaseelo-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1574,"normalizedScore":87.8738,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaautomationbench-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.6,"normalizedScore":79.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaenterpriseopsgym-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaharveylab-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.6,"normalizedScore":98.7608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aatau3banking-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":65.2632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-terminalbenchhard-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":62.9,"normalizedScore":92.9245,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaterminalbench21-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.6,"normalizedScore":86.7647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-sweverified-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":95,"normalizedScore":98.6188,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-swepro-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":80,"normalizedScore":99.1979,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-frontiercode-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":53.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-vulcanbench-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vulcanbench","benchmarkName":"VulcanBench v3","benchmarkCategory":"coding","benchmarkOrganisation":"VulcanBench contributors","benchmarkVersion":"2026","score":87,"normalizedScore":75.2731,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aacodingindex-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76.49,"normalizedScore":97.894,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aascicode-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":60.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-lcr-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70,"normalizedScore":92.4703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-critpt-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":28.6,"normalizedScore":88.5449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-blueprintbench2-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"blueprint-bench-2","benchmarkName":"Blueprint-Bench 2","benchmarkCategory":"multimodal","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":38.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-officeqapro-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":57.9,"normalizedScore":61.3734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-designarenawebsite-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1324,"normalizedScore":91.196,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aagpqadiamond-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":92.6,"normalizedScore":97.8693,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aahle-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":53.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaomniscienceindex-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.2,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-omniscienceaccuracy-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-omnisciencehallucinationrate-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.9,"normalizedScore":50.7841,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-fable-aaifbench-2026-08-01","modelSlug":"claude-fable-5","modelName":"Claude Fable 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":63.5,"normalizedScore":71.3864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Fable 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-terminalbench2-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":74.6,"normalizedScore":69.2171,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-browsecomp-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":84.3,"normalizedScore":83.4728,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-deepsearchqa-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":93.1,"normalizedScore":94.0994,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-osworldverified-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":83.4,"normalizedScore":96.5217,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-financeagentv2-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":53.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gdpvalaa-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1593,"normalizedScore":86.421,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-mcpatlas-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":82.2,"normalizedScore":89.9317,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-toolathlon-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":59.9,"normalizedScore":67.7618,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gertlabs-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":72.97,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaagenticindex-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.18,"normalizedScore":85.3064,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-tau2bench-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.4,"normalizedScore":95.2573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gdpvalaanormalized-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.6,"normalizedScore":80.1762,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-researchclawbench-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":21.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-osworld2-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":20.6,"normalizedScore":26.2537,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aabriefcaseelo-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1346,"normalizedScore":68.9369,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaautomationbench-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.5,"normalizedScore":79,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaenterpriseopsgym-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44,"normalizedScore":72.2656,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaharveylab-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.1,"normalizedScore":95.6629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aatau3banking-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.6,"normalizedScore":69.4737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-terminalbenchhard-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":58.3,"normalizedScore":82.0755,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaterminalbench21-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.6,"normalizedScore":86.7647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-sweverified-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":88.6,"normalizedScore":89.779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-swepro-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":69.2,"normalizedScore":70.3209,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-swemultilingual-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":84.4,"normalizedScore":80.597,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-swemultimodal-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultimodal","benchmarkName":"SWE-bench Multimodal","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":38.4,"normalizedScore":32.9073,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aacodingindex-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.25,"normalizedScore":94.7279,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aascicode-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.5,"normalizedScore":88.7015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-frontiercode-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":46.5,"normalizedScore":76.0274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-arcagi2-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":72.08,"normalizedScore":74.1191,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-arcagi3-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":1.52,"normalizedScore":4.7556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-lcr-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.7,"normalizedScore":89.432,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-critpt-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":20.9,"normalizedScore":64.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-officeqapro-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":66.2,"normalizedScore":96.9957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-screenspotpro-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":87.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-charxiv-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":89.9,"normalizedScore":91.1765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-charxivnotools-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":80.5,"normalizedScore":29.4118,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-designarenawebsite-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1268,"normalizedScore":81.8937,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gpqa-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-gpqadiamond-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-hle-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":57.9,"normalizedScore":88.0702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-hlenotools-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":49.8,"normalizedScore":82.8996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaomniscienceindex-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.4,"normalizedScore":89.9529,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-omniscienceaccuracy-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.6,"normalizedScore":74.5704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-omnisciencehallucinationrate-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.9,"normalizedScore":73.7033,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-include-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-include","benchmarkName":"INCLUDE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.6,"normalizedScore":67.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-aaifbench-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":62.2,"normalizedScore":69.469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-usamo2026-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":96.7,"normalizedScore":92.4306,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-frontiermathv2tiers13-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":47.241,"normalizedScore":53.0798,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-8-frontiermathv2tier4-2026-08-01","modelSlug":"claude-opus-4-8","modelName":"Claude Opus 4.8","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":31.25,"normalizedScore":37.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.8; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-terminalbench2-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":69.7,"normalizedScore":60.4982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-qwenclawbench-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":64.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-qwenwebbench-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1568,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-claweval-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":65.2,"normalizedScore":83.3799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-bfclv4-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mcpatlas-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":76.4,"normalizedScore":80.0341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-vitabench-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":47.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-hlewithtools-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":53.5,"normalizedScore":58.9744,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaagenticindex-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.59,"normalizedScore":55.1373,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-tau2bench-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.7,"normalizedScore":95.56,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gdpvalaanormalized-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.6,"normalizedScore":56.6814,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gdpvalaa-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1271,"normalizedScore":70.1666,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gertlabs-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":64.27,"normalizedScore":81.6145,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-researchclawbench-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18.7,"normalizedScore":72.4138,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aabriefcaseelo-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":912,"normalizedScore":32.8904,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaenterpriseopsgym-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45,"normalizedScore":76.1719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaitbench-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.5,"normalizedScore":72.9249,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-terminalbenchhard-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":50.8,"normalizedScore":64.3868,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaterminalbench21-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.5,"normalizedScore":57.0588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaharveylab-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.4,"normalizedScore":86.1214,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-sweverified-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.4,"normalizedScore":78.453,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-swepro-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":60.6,"normalizedScore":47.3262,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-swemultilingual-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.3,"normalizedScore":65.4229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-nl2repo-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":47.2,"normalizedScore":74.0741,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-scicode-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":53.5,"normalizedScore":80.0604,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-livecodebench-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":91.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aacodingindex-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.97,"normalizedScore":83.0247,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aascicode-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":48.8,"normalizedScore":80.7757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mrcrv2-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":90.4,"normalizedScore":93.6255,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-critpt-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":13.4,"normalizedScore":41.4861,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-lcr-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69,"normalizedScore":91.1493,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-designarenawebsite-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1303,"normalizedScore":87.7076,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gpqa-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.4,"normalizedScore":95.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-gpqadiamond-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.4,"normalizedScore":95.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-hle-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":41.4,"normalizedScore":59.1228,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmlupro-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":89.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmluredux-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":95,"normalizedScore":93.9714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-supergpqa-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":73.6,"normalizedScore":70.2199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmmlu-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90.3,"normalizedScore":92,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaomniscienceindex-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.1,"normalizedScore":79.5133,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-omniscienceaccuracy-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.1,"normalizedScore":46.2199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-omnisciencehallucinationrate-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.9,"normalizedScore":89.3848,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-mmluprox-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":87,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-nova63-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59,"normalizedScore":97.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-include-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-include","benchmarkName":"INCLUDE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":47.0588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-maxife-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-maxife","benchmarkName":"MAXIFE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":89.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-polymath-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-polymath","benchmarkName":"PolyMath","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-ifeval-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.3,"normalizedScore":97.9314,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-ifbench-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":79.1,"normalizedScore":87.3391,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-aaifbench-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":80.5,"normalizedScore":96.4602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-hmmtfeb2026-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":97.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-imoanswerbench-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-max-apex-2026-08-01","modelSlug":"qwen-3-7-max","modelName":"Qwen3.7-Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":44.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-terminalbench2-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":87.4,"normalizedScore":91.9929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-browsecomp-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":87.5,"normalizedScore":90.1674,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-osworld2-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":50.2,"normalizedScore":69.9115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-cybergym-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":81.8,"normalizedScore":88.3295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-exploitgym-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":23.2,"normalizedScore":68.0851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-toolathlon-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":53.1,"normalizedScore":53.7988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaagenticindex-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.38,"normalizedScore":85.6701,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-tau2bench-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86.3,"normalizedScore":87.0838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.1,"normalizedScore":79.442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gdpvalaa-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1583,"normalizedScore":85.9162,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaharveylab-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.2,"normalizedScore":88.3519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaitbench-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51,"normalizedScore":89.7233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aatau3banking-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.8,"normalizedScore":91.5789,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaautomationbench-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.6,"normalizedScore":64.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-terminalbenchhard-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":57.6,"normalizedScore":80.4245,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaterminalbench21-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88,"normalizedScore":96.7647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-apexagentsaa-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":38.9,"normalizedScore":82.3276,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-swepro-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":63.4,"normalizedScore":54.8128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-deepswe-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":69.6,"normalizedScore":90.4025,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiercode11extended-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode11extended","benchmarkName":"FrontierCode 1.1 Extended","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":55.8,"normalizedScore":8.2353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aacodingindex-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76.66,"normalizedScore":98.1343,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aascicode-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.9,"normalizedScore":89.3761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-arcagi2-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":83.9,"normalizedScore":89.1001,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-arcagi3-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.8,"normalizedScore":2.3612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-lcr-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-critpt-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":30,"normalizedScore":92.8793,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-mmmupro-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":80.7,"normalizedScore":57.0968,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-mmmupropython-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":82,"normalizedScore":82.7815,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aammmupro-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.7,"normalizedScore":93.1271,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gpqa-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.9,"normalizedScore":96.2905,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-gpqadiamond-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.9,"normalizedScore":96.2905,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-healthbenchprofessional-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":57.7,"normalizedScore":77.4194,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-healthbenchhard-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":32.7,"normalizedScore":63.9286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aahle-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":41.8,"normalizedScore":77.0916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-0.2,"normalizedScore":68.2889,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.9,"normalizedScore":73.3677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.2,"normalizedScore":14.234,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-aaifbench-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":71.2,"normalizedScore":82.7434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiermath-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":84.9,"normalizedScore":90.9292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":84.9,"normalizedScore":95.3933,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-frontiermathv2tier4-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":68.3,"normalizedScore":82.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-terra-exploitbench-2026-08-01","modelSlug":"gpt-5-6-terra","modelName":"GPT-5.6 Terra","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-exploitbench","benchmarkName":"ExploitBench v8-bench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Seunghyun Lee, David Brumley, Carnegie Mellon University","benchmarkVersion":"2026","score":52.9,"normalizedScore":50.3614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Terra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-terminalbench2-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":80.4,"normalizedScore":79.5374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-browsecomp-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":84.7,"normalizedScore":84.3096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-hlewithtools-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":57.4,"normalizedScore":73.2601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-osworldverified-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":81.2,"normalizedScore":91.7391,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-gdpvalaa-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1603,"normalizedScore":86.9258,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaagenticindex-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.69,"normalizedScore":84.4153,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-gdpvalaanormalized-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.2,"normalizedScore":81.0573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aabriefcaseelo-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1386,"normalizedScore":72.2591,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaautomationbench-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.2,"normalizedScore":32.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaenterpriseopsgym-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.7,"normalizedScore":75,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaharveylab-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.1,"normalizedScore":94.4238,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aatau3banking-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.2,"normalizedScore":72.6316,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaterminalbench21-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.5,"normalizedScore":74.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-sweverified-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":85.2,"normalizedScore":85.0829,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-swepro-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":63.2,"normalizedScore":54.2781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-swemultilingual-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.3,"normalizedScore":65.4229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-swemultimodal-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultimodal","benchmarkName":"SWE-bench Multimodal","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":28.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-frontiercode-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":42.7,"normalizedScore":63.0137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aacodingindex-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.55,"normalizedScore":90.9117,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aascicode-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.6,"normalizedScore":88.8702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-lcr-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70.7,"normalizedScore":93.395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-critpt-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":16.9,"normalizedScore":52.322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-charxiv-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":88.3,"normalizedScore":87.2549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-charxivnotools-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":77,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aammmupro-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":77.3,"normalizedScore":87.2852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-designarenawebsite-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1314,"normalizedScore":89.5349,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-hle-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":57.4,"normalizedScore":87.193,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-hlenotools-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":43.2,"normalizedScore":70.632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aagpqadiamond-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":91.1,"normalizedScore":95.7386,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-aaomniscienceindex-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.3,"normalizedScore":80.4553,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-omniscienceaccuracy-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.3,"normalizedScore":60.3093,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-5-omnisciencehallucinationrate-2026-08-01","modelSlug":"claude-sonnet-5","modelName":"Claude Sonnet 5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.3,"normalizedScore":72.0145,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-terminalbench2-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":70.3,"normalizedScore":61.5658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-qwenclawbench-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":61.8,"normalizedScore":80,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-qwenwebbench-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1536,"normalizedScore":81.2865,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-claweval-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.7,"normalizedScore":79.8883,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-bfclv4-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":72.9,"normalizedScore":96.1089,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mcpatlas-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":73.2,"normalizedScore":74.5734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-vitabench-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":45.6,"normalizedScore":92.9012,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-deepplanning-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":62.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-osworldverified-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":73.3,"normalizedScore":74.5652,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-androidworld-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":81,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aaagenticindex-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.81,"normalizedScore":37.3522,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-apexagentsaa-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":22.4,"normalizedScore":46.7672,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-tau2bench-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93,"normalizedScore":93.8446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gdpvalaanormalized-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.1,"normalizedScore":32.4523,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gdpvalaa-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":943,"normalizedScore":53.6093,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-osworld2-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":2.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-sweverified-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.7,"normalizedScore":74.7238,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-swepro-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.6,"normalizedScore":39.3048,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-swemultilingual-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":75.8,"normalizedScore":59.204,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-nl2repo-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":41.1,"normalizedScore":51.4815,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-scicode-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":51.3,"normalizedScore":73.4139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-livecodebench-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":89.6,"normalizedScore":96.2963,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aacodingindex-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.86,"normalizedScore":68.735,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aascicode-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.5,"normalizedScore":75.2108,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-critpt-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":9.1,"normalizedScore":28.1734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mrcrv2-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":91.7,"normalizedScore":96.2151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-lcr-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65,"normalizedScore":85.8653,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmmupro-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79,"normalizedScore":51.6129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mathvision-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":90.3,"normalizedScore":80,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-charxiv-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":85.9,"normalizedScore":81.3725,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-erqa-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":69.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-medxpertqamm-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":71,"normalizedScore":68.4049,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-screenspotpro-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":79,"normalizedScore":78.91,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-simplevqa-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":81.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmsearchplus-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmsearchplus","benchmarkName":"MMSearch-Plus","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":41.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-realworldqa-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-omnidocbench15-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnidocbench15","benchmarkName":"OmniDocBench 1.5","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":91.4,"normalizedScore":88.2353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-ocrbenchv2-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ocrbench-v2","benchmarkName":"OCRBench v2","benchmarkCategory":"multimodal","benchmarkOrganisation":"OCRBench authors","benchmarkVersion":"2025","score":70.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-odinw13-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-odinw13","benchmarkName":"ODINW13","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":51.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-videommewithsub-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-videommmu-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":85.4,"normalizedScore":43.5897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mlvuavg-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mlvuavg","benchmarkName":"MLVU mean average","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aammmupro-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.5,"normalizedScore":92.7835,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-designarenawebsite-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1297,"normalizedScore":86.711,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gpqa-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.3,"normalizedScore":92.581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-gpqadiamond-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":90.3,"normalizedScore":92.581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-hle-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":34.7,"normalizedScore":47.3684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmlupro-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":88.5,"normalizedScore":98.4348,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmluredux-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.5,"normalizedScore":92.0874,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-supergpqa-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":71.4,"normalizedScore":67.1584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmmlu-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":89,"normalizedScore":74.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aaomniscienceindex-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.4,"normalizedScore":70.3297,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-omniscienceaccuracy-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.2,"normalizedScore":32.646,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-omnisciencehallucinationrate-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.5,"normalizedScore":86.2485,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-mmluprox-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":85.4,"normalizedScore":78.9474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-nova63-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":58.8,"normalizedScore":92.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-include-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-include","benchmarkName":"INCLUDE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-maxife-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-maxife","benchmarkName":"MAXIFE","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-polymath-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-polymath","benchmarkName":"PolyMath","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-ifeval-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.6,"normalizedScore":98.818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-ifbench-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":79.1,"normalizedScore":87.3391,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-aaifbench-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":78,"normalizedScore":92.7729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-hmmtfeb2026-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.9,"normalizedScore":94.1127,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-imoanswerbench-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":86,"normalizedScore":92.6874,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-7-plus-apex-2026-08-01","modelSlug":"qwen-3-7-plus","modelName":"Qwen3.7-Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":22.7,"normalizedScore":50.5669,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.7 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-terminalbench2-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":80,"normalizedScore":78.8256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-mcpatlas-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":88.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-toolathlon-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":75.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-osworldverified-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":80.8,"normalizedScore":90.8696,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-webarenaverified-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-webarenaverified","benchmarkName":"WebArena-Verified Browser Agent Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Amine El Hattami, Megh Thakkar, Nicolas Chapados, Christopher Pal","benchmarkVersion":"2025","score":69,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-deepsearchqa-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":84.9,"normalizedScore":68.6335,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-cybergym-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":59,"normalizedScore":36.1556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-financeagentv2-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":57.2,"normalizedScore":83.3123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-deepswe-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":53.3,"normalizedScore":39.9381,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-osworld2-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":14.2,"normalizedScore":16.8142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-jobbench-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":54.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-cybench-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"cybench","benchmarkName":"Cybench","benchmarkCategory":"agents","benchmarkOrganisation":"Stanford / Cybench authors","benchmarkVersion":"2025","score":92.9,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-exploitgym-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":0.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaagenticindex-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.54,"normalizedScore":67.776,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-gdpvalaanormalized-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.8,"normalizedScore":64.3172,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-gdpvalaa-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1375,"normalizedScore":75.4165,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aabriefcaseelo-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":868,"normalizedScore":29.2359,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaautomationbench-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.8,"normalizedScore":50.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaharveylab-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.1,"normalizedScore":98.1413,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aatau3banking-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.2,"normalizedScore":56.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaterminalbench21-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.9,"normalizedScore":67.0588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaenterpriseopsgym-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.2,"normalizedScore":84.7656,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-swepro-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":61.5,"normalizedScore":49.7326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aacodingindex-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.34,"normalizedScore":90.6148,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aascicode-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":58.2,"normalizedScore":96.6273,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-mrcr1m-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":54.1,"normalizedScore":48.3304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-lcr-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.3,"normalizedScore":83.6196,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-critpt-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":15.1,"normalizedScore":46.7492,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-charxiv-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":88.4,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-babyvision-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-babyvision","benchmarkName":"BabyVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":76.3,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-designarenawebsite-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1291,"normalizedScore":85.7143,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-hle-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":62.1,"normalizedScore":95.4386,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-hlenotools-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":52.2,"normalizedScore":87.3606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-healthbenchprofessional-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":59.3,"normalizedScore":90.3226,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aagpqadiamond-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.8,"normalizedScore":93.892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-aaomniscienceindex-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18,"normalizedScore":82.5746,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-omniscienceaccuracy-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.6,"normalizedScore":64.2612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-1-1-omnisciencehallucinationrate-2026-08-01","modelSlug":"muse-spark-1-1","modelName":"Muse Spark 1.1","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.1,"normalizedScore":71.0495,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Muse Spark 1.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-terminalbench2-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":82,"normalizedScore":82.3843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-cybergym-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":81.8,"normalizedScore":88.3295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-browsecomp-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":84.4,"normalizedScore":83.682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-osworldverified-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":78.7,"normalizedScore":86.3043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mcpatlas-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":75.3,"normalizedScore":78.157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-toolathlon-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":55.6,"normalizedScore":58.9322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-tau2bench-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98,"normalizedScore":98.89,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaagenticindex-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.87,"normalizedScore":81.1057,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-apexagentsaa-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":37.7,"normalizedScore":79.7414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.6,"normalizedScore":72.8341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gdpvalaa-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1491,"normalizedScore":81.2721,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gertlabs-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":72.93,"normalizedScore":99.9155,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-researchclawbench-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":17,"normalizedScore":52.8736,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-osworld2-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":13,"normalizedScore":15.0442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-jobbench-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":42.7,"normalizedScore":74.0091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-exploitgym-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":13.4,"normalizedScore":38.2979,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaautomationbench-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.1,"normalizedScore":47,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaenterpriseopsgym-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.6,"normalizedScore":82.4219,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaitbench-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.8,"normalizedScore":79.4466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-swepro-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":58.6,"normalizedScore":41.9786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-vibecodebench-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":69.847,"normalizedScore":98.3719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-reactnativeevals-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":84.7,"normalizedScore":54.5817,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aacodingindex-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.89,"normalizedScore":95.6325,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aascicode-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":93.086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiercode-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":43,"normalizedScore":64.0411,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mrcrv2-64-128-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mrcrv2-64-128","benchmarkName":"OpenAI MRCR v2 8-needle 64K-128K","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mrcrv2-128-256-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mrcrv2-128-256","benchmarkName":"OpenAI MRCR v2 8-needle 128K-256K","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":87.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-arcagi2-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":85,"normalizedScore":90.4943,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-arcagi3-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.43,"normalizedScore":1.1307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-lcr-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.3,"normalizedScore":98.1506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-critpt-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":27.1,"normalizedScore":83.9009,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mmmupro-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81.2,"normalizedScore":58.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-mmmupropython-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.2,"normalizedScore":90.7285,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-officeqapro-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":54.1,"normalizedScore":45.0644,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aammmupro-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":79.9,"normalizedScore":91.7526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-designarenawebsite-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1280,"normalizedScore":83.887,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gpqa-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-gpqadiamond-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":93.6,"normalizedScore":97.2892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-hle-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":52.2,"normalizedScore":78.0702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-hlenotools-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":41.4,"normalizedScore":67.2862,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.1,"normalizedScore":84.2229,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.9,"normalizedScore":92.268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.5,"normalizedScore":13.8721,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-aaifbench-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.9,"normalizedScore":89.6755,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiermath-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":51.7,"normalizedScore":17.4779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":51.7,"normalizedScore":58.0899,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-5-frontiermathv2tier4-2026-08-01","modelSlug":"gpt-5-5","modelName":"GPT-5.5","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":35.4,"normalizedScore":42.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-pro-claweval-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.8,"normalizedScore":73.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-deepsearchqa-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":69.7,"normalizedScore":21.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-tau2bench-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.6,"normalizedScore":96.4682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaagenticindex-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.4,"normalizedScore":38.4252,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-apexagentsaa-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":32,"normalizedScore":67.4569,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gdpvalaanormalized-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.2,"normalizedScore":34.0675,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gdpvalaa-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":965,"normalizedScore":54.7198,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gertlabs-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":56.87,"normalizedScore":65.9763,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-researchclawbench-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":13.3,"normalizedScore":10.3448,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaautomationbench-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.5,"normalizedScore":24,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-livecodebenchpro-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":82.9,"normalizedScore":88.4028,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-reactnativeevals-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":78.9,"normalizedScore":31.4741,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-vibecodebench-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":32.034,"normalizedScore":45.1164,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aacodingindex-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.83,"normalizedScore":87.0671,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aascicode-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":58.9,"normalizedScore":97.8078,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-arcagi2-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":77.08,"normalizedScore":80.4563,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-arcagi3-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.42,"normalizedScore":1.0974,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-lcr-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.7,"normalizedScore":96.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-critpt-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":17.7,"normalizedScore":54.7988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-mmmupro-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":83.9,"normalizedScore":67.4194,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-charxiv-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":80.2,"normalizedScore":67.402,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-erqa-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":69.4,"normalizedScore":97.8022,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-simplevqa-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":72.4,"normalizedScore":63.6719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-screenspotpro-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":84.4,"normalizedScore":91.7062,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-zerobench-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"2026","score":29,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-medxpertqamm-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":81.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aammmupro-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":82.4,"normalizedScore":96.0481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-designarenawebsite-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1276,"normalizedScore":83.2226,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-gpqadiamond-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":94.3,"normalizedScore":98.2879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-hlenotools-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":45.4,"normalizedScore":74.7212,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-healthbenchhard-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":20.6,"normalizedScore":20.7143,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-medxpertqatext-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":71.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aahle-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":44.7,"normalizedScore":82.8685,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaomniscienceindex-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.9,"normalizedScore":94.27,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-omniscienceaccuracy-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.3,"normalizedScore":89.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.9,"normalizedScore":56.8154,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaglobalmmlulite-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-aaifbench-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":77.1,"normalizedScore":91.4454,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-frontiermathv2tiers13-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":36.9,"normalizedScore":41.4607,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-1-pro-frontiermathv2tier4-2026-08-01","modelSlug":"gemini-3-1-pro","modelName":"Gemini 3.1 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":16.7,"normalizedScore":20.1205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gpt-5-6-luna-terminalbench2-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":84.7,"normalizedScore":87.1886,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-browsecomp-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.3,"normalizedScore":81.3808,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-osworld2-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":45.6,"normalizedScore":63.1268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-cybergym-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":77.9,"normalizedScore":79.405,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-exploitgym-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":12.4,"normalizedScore":35.2584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-toolathlon-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":53.4,"normalizedScore":54.4148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaagenticindex-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.6,"normalizedScore":82.4332,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.1,"normalizedScore":79.442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gdpvalaa-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1582,"normalizedScore":85.8657,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaharveylab-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.9,"normalizedScore":91.6976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaitbench-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.3,"normalizedScore":68.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aatau3banking-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.2,"normalizedScore":67.3684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaautomationbench-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.2,"normalizedScore":47.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaterminalbench21-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.9,"normalizedScore":75.8824,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-apexagentsaa-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":35.8,"normalizedScore":75.6466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-swepro-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":62.7,"normalizedScore":52.9412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-deepswe-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":67.2,"normalizedScore":82.9721,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiercode11extended-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode11extended","benchmarkName":"FrontierCode 1.1 Extended","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":55.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aacodingindex-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.45,"normalizedScore":90.7703,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aascicode-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":52.5,"normalizedScore":87.0152,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-arcagi2-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":59.54,"normalizedScore":58.2256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-arcagi3-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.18,"normalizedScore":0.2993,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-lcr-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-critpt-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":20.6,"normalizedScore":63.7771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-mmmupro-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.4,"normalizedScore":49.6774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-mmmupropython-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":79.5,"normalizedScore":66.2252,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aammmupro-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.6,"normalizedScore":89.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gpqa-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.3,"normalizedScore":95.4344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-gpqadiamond-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.3,"normalizedScore":95.4344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-healthbenchprofessional-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":55.7,"normalizedScore":61.2903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-healthbenchhard-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":32,"normalizedScore":61.4286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aahle-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":37.2,"normalizedScore":67.9283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-11.2,"normalizedScore":59.6546,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.5,"normalizedScore":65.8076,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.1,"normalizedScore":8.3233,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiermath-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"frontiermath","benchmarkName":"FrontierMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2024","score":78.6,"normalizedScore":76.9912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":78.6,"normalizedScore":88.3146,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-frontiermathv2tier4-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":58.5,"normalizedScore":70.4819,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-6-luna-exploitbench-2026-08-01","modelSlug":"gpt-5-6-luna","modelName":"GPT-5.6 Luna","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-exploitbench","benchmarkName":"ExploitBench v8-bench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Seunghyun Lee, David Brumley, Carnegie Mellon University","benchmarkVersion":"2026","score":33.2,"normalizedScore":2.8916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.6 Luna; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-terminalbench2-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":81,"normalizedScore":80.605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-mcpatlas-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":76.8,"normalizedScore":80.7167,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-toolathlon-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":48.2,"normalizedScore":43.7372,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaagenticindex-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.06,"normalizedScore":77.8141,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-tau2bench-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":99.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gdpvalaanormalized-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.5,"normalizedScore":74.1557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gdpvalaa-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1510,"normalizedScore":82.2312,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-apexagentsaa-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":33.7,"normalizedScore":71.1207,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-researchclawbench-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":20.7,"normalizedScore":95.4023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aabriefcaseelo-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1254,"normalizedScore":61.2957,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaenterpriseopsgym-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.7,"normalizedScore":67.1875,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaharveylab-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91,"normalizedScore":95.539,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaitbench-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.7,"normalizedScore":73.3202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aatau3banking-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":65.2632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-terminalbenchhard-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":50.8,"normalizedScore":64.3868,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaterminalbench21-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.9,"normalizedScore":67.0588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-swepro-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":62.1,"normalizedScore":51.3369,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-nl2repo-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":48.9,"normalizedScore":80.3704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-programbench-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-programbench","benchmarkName":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","benchmarkCategory":"coding","benchmarkOrganisation":"John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","benchmarkVersion":"2026","score":63.7,"normalizedScore":25.6345,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aacodingindex-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.76,"normalizedScore":86.9682,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aascicode-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50.5,"normalizedScore":83.6425,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-critpt-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":20.9,"normalizedScore":64.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-lcr-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.3,"normalizedScore":94.1876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-designarenawebsite-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1341,"normalizedScore":94.0199,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gpqa-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":91.2,"normalizedScore":93.865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-gpqadiamond-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":91.2,"normalizedScore":93.865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hle-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":54.7,"normalizedScore":82.4561,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hlenotools-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":40.5,"normalizedScore":65.6134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaomniscienceindex-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4,"normalizedScore":71.5856,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-omniscienceaccuracy-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.1,"normalizedScore":37.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-omnisciencehallucinationrate-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.1,"normalizedScore":83.1122,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaopennessindex-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.4,"normalizedScore":22.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aaifbench-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.3,"normalizedScore":85.8407,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-aime2026-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":99.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hmmtnov2025-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":94.4,"normalizedScore":67.9487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-hmmtfeb2026-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.5,"normalizedScore":93.552,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-2-mmanswerbench-2026-08-01","modelSlug":"glm-5-2","modelName":"GLM-5.2","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":91,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-terminalbench2-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":76.2,"normalizedScore":72.0641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mcpatlas-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":83.6,"normalizedScore":92.3208,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-toolathlon-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":56.5,"normalizedScore":60.7803,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-osworldverified-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":78.4,"normalizedScore":85.6522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-financeagentv2-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"finance-agent-v2","benchmarkName":"Finance Agent v2","benchmarkCategory":"research","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":57.861,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gdpvalaa-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1345,"normalizedScore":73.9021,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-tau2bench-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.3,"normalizedScore":96.1655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gdpvalaanormalized-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.2,"normalizedScore":61.9677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaagenticindex-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.45,"normalizedScore":67.6123,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-apexagentsaa-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":47.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gertlabs-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":61.85,"normalizedScore":76.5004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-researchclawbench-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18,"normalizedScore":64.3678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaautomationbench-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.6,"normalizedScore":49.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaenterpriseopsgym-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.1,"normalizedScore":96.0938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-swepro-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.1,"normalizedScore":32.6203,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-scicode-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":53.1,"normalizedScore":78.852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-vibecodebench-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":48.683,"normalizedScore":68.5647,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aacodingindex-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70.14,"normalizedScore":88.9187,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aascicode-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.1,"normalizedScore":88.027,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mrcrv2-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":77.3,"normalizedScore":67.5299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mrcr1m-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":26.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-arcagi2-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":72.1,"normalizedScore":74.1445,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lcr-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":91.5456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-critpt-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":13.1,"normalizedScore":40.5573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-charxiv-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":84.2,"normalizedScore":77.2059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-mmmupro-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":83.6,"normalizedScore":66.4516,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-blueprintbench2-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"blueprint-bench-2","benchmarkName":"Blueprint-Bench 2","benchmarkCategory":"multimodal","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":33.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aammmupro-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":84.3,"normalizedScore":99.3127,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-designarenawebsite-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1283,"normalizedScore":84.3854,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gpqa-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.2,"normalizedScore":95.2918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-gpqadiamond-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.676,"normalizedScore":95.9709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-hle-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":40.2,"normalizedScore":57.0175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-omniscienceaccuracy-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.9,"normalizedScore":83.677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.7,"normalizedScore":43.7877,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaomniscienceindex-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.7,"normalizedScore":86.2637,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-ifbench-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":76.3,"normalizedScore":81.3305,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-aaifbench-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.3,"normalizedScore":90.2655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-frontiermathv2tiers13-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":38.966,"normalizedScore":43.782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-frontiermathv2tier4-2026-08-01","modelSlug":"gemini-3-5-flash","modelName":"Gemini 3.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.583,"normalizedScore":17.5699,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-terminalbench2-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":65.4,"normalizedScore":52.847,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-browsecomp-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.7,"normalizedScore":82.2176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-osworldverified-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":72.7,"normalizedScore":73.2609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-tau2bench-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-claweval-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":70.4,"normalizedScore":90.6425,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-deepsearchqa-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":73.7,"normalizedScore":33.8509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-cybergym-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":66.6,"normalizedScore":53.5469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-gertlabs-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":61.85,"normalizedScore":76.5004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-researchclawbench-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":19.9,"normalizedScore":86.2069,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-jobbench-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":36.7,"normalizedScore":61.0136,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-sweverified-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.84,"normalizedScore":79.0608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-sweverifiedarcee-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-livecodebenchpro-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":70.7,"normalizedScore":70.4932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-swepro-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":53.4,"normalizedScore":28.0749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-swerebench-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":65.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-reactnativeevals-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":84.1,"normalizedScore":52.1912,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-vibecodebench-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":57.573,"normalizedScore":81.0853,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aascicode-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.7,"normalizedScore":75.5481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-frontiercode-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":26.9,"normalizedScore":8.9041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-lcr-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.3,"normalizedScore":77.0145,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-critpt-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.8,"normalizedScore":8.6687,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-mmmupro-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":77.3,"normalizedScore":46.129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-erqa-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":51.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-screenspotpro-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":83.1,"normalizedScore":88.6256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-medxpertqamm-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":64.8,"normalizedScore":49.3865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aammmupro-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.5,"normalizedScore":79.0378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-designarenawebsite-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1319,"normalizedScore":90.3654,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-gpqa-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":91.3,"normalizedScore":94.0077,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-gpqadiamond-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":89.2,"normalizedScore":91.0116,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-supergpqa-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-mmlupro-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":82,"normalizedScore":89.1861,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-mmluproarcee-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":89.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-hle-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":53,"normalizedScore":79.4737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-hlenotools-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":40,"normalizedScore":64.684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-healthbenchhard-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":14.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-medxpertqatext-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":52.1,"normalizedScore":8.9202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aaomniscienceindex-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.5,"normalizedScore":71.1931,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-omniscienceaccuracy-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.2,"normalizedScore":72.1649,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-omnisciencehallucinationrate-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":76,"normalizedScore":25.3317,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aaifbench-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":44.6,"normalizedScore":43.5103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-aime2025arcee-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":99.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-frontiermathv2tiers13-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":40.7,"normalizedScore":45.7303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-6-frontiermathv2tier4-2026-08-01","modelSlug":"claude-opus-4-6","modelName":"Claude Opus 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":22.9,"normalizedScore":27.5904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-terminalbench2-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":75.1,"normalizedScore":70.1068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-cybergym-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":79,"normalizedScore":81.9222,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-browsecomp-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":82.7,"normalizedScore":80.1255,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-osworldverified-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":75,"normalizedScore":78.2609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mcpatlas-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":70.6,"normalizedScore":70.1365,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-toolathlon-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":54.6,"normalizedScore":56.8789,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-tau2bench-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.9,"normalizedScore":99.7982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-claweval-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":60.3,"normalizedScore":76.5363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-deepsearchqa-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":73.6,"normalizedScore":33.5404,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aaagenticindex-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.08,"normalizedScore":74.2135,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-apexagentsaa-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":33.3,"normalizedScore":70.2586,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.6,"normalizedScore":65.4919,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gdpvalaa-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1392,"normalizedScore":76.2746,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gertlabs-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":64.89,"normalizedScore":82.9248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-researchclawbench-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":15.3,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-jobbench-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":38.9,"normalizedScore":65.7786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-exploitgym-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"exploitgym","benchmarkName":"ExploitGym","benchmarkCategory":"agents","benchmarkOrganisation":"ExploitGym authors","benchmarkVersion":"2026","score":6,"normalizedScore":15.8055,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-livecodebenchpro-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":87.5,"normalizedScore":95.1556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-swepro-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.7,"normalizedScore":39.5722,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-reactnativeevals-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":85.3,"normalizedScore":56.9721,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-vibecodebench-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":67.421,"normalizedScore":94.9551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aacodingindex-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":71.05,"normalizedScore":90.2049,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aascicode-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.6,"normalizedScore":93.9292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-arcagi2-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":73.95,"normalizedScore":76.4892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-arcagi3-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.21,"normalizedScore":0.3991,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-lcr-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-critpt-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":23.4,"normalizedScore":72.4458,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mmmupro-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81.2,"normalizedScore":58.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-officeqapro-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":53.2,"normalizedScore":41.2017,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mmmupropython-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":82.1,"normalizedScore":83.4437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-charxiv-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":82.8,"normalizedScore":73.7745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-erqa-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":65.4,"normalizedScore":75.8242,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-simplevqa-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":61.1,"normalizedScore":19.5313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-screenspotpro-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":85.4,"normalizedScore":94.0758,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-zerobench-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"2026","score":41,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-medxpertqamm-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":77.1,"normalizedScore":87.1166,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aammmupro-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.4,"normalizedScore":89.1753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-designarenawebsite-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1245,"normalizedScore":78.0731,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gpqa-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.8,"normalizedScore":96.1478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-hle-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":52.1,"normalizedScore":77.8947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-hlenotools-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":39.8,"normalizedScore":64.3123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-gpqadiamond-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":92.8,"normalizedScore":96.1478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-healthbenchhard-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":40.1,"normalizedScore":90.3571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-medxpertqatext-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":59.6,"normalizedScore":44.1315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.7,"normalizedScore":72.9199,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50,"normalizedScore":80.4124,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.6,"normalizedScore":10.1327,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-healthbenchprofessional-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-healthbenchprofessional","benchmarkName":"HealthBench Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Rebecca Soskin Hicks, Mikhail Trofimov, Dominick Lim, Rahul K. Arora, Foivos Tsimpourlas, Preston Bowman, Michael Sharman, Chi Tong, Kavin Karthik, Arnav Dugar, Akshay Jagadeesh, Khaled Saab, Johannes Heidecke, Ashley Alexander, Nate Gross, Karan Singhal","benchmarkVersion":"2026","score":48.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-aaifbench-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.9,"normalizedScore":86.7257,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":47.6,"normalizedScore":53.4831,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-frontiermathv2tier4-2026-08-01","modelSlug":"gpt-5-4","modelName":"GPT-5.4","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":27.1,"normalizedScore":32.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-terminalbench2-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":67.9,"normalizedScore":57.2954,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-browsecomp-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.4,"normalizedScore":81.59,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-hlewithtools-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":48.2,"normalizedScore":39.5604,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-mcpatlas-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":73.6,"normalizedScore":75.256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gdpvalaa-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1306,"normalizedScore":71.9334,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-toolathlon-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":51.8,"normalizedScore":51.1294,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaagenticindex-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":36.36,"normalizedScore":65.6301,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-apexagentsaa-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":24.3,"normalizedScore":50.8621,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-tau2bench-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":96.2,"normalizedScore":97.0737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gdpvalaanormalized-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.3,"normalizedScore":59.1777,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aabriefcaseelo-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":930,"normalizedScore":34.3854,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaenterpriseopsgym-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.4,"normalizedScore":58.2031,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaharveylab-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.4,"normalizedScore":87.3606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaitbench-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.3,"normalizedScore":64.6245,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aatau3banking-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.8,"normalizedScore":60,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-terminalbenchhard-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":46.2,"normalizedScore":53.5377,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaterminalbench21-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64,"normalizedScore":26.1765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-livecodebenchpass1cot-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-livecodebenchpass1cot","benchmarkName":"LiveCodeBench Pass@1 with Chain-of-Thought","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek","benchmarkVersion":"2026","score":93.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-codeforces-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-codeforces","benchmarkName":"Codeforces Rating","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":3206,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-sweverified-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.6,"normalizedScore":78.7293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-swepro-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.4,"normalizedScore":33.4225,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-swemultilingual-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":76.2,"normalizedScore":60.199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-vibecodebench-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":49.931,"normalizedScore":70.3224,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aacodingindex-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59.36,"normalizedScore":73.682,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aascicode-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50,"normalizedScore":82.7993,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-mrcr1m-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":83.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-corpusqa1m-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":62,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-lcr-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.3,"normalizedScore":87.5826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-critpt-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":12.9,"normalizedScore":39.9381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-designarenawebsite-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1260,"normalizedScore":80.5648,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-mmlupro-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":87.5,"normalizedScore":97.012,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-simpleqa-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":57.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-chinesesimpleqa-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":84.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gpqa-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.1,"normalizedScore":92.2956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-gpqadiamond-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":90.1,"normalizedScore":92.2956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-hle-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":37.7,"normalizedScore":52.6316,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaomniscienceindex-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10,"normalizedScore":60.5965,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-omniscienceaccuracy-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.3,"normalizedScore":68.9003,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-omnisciencehallucinationrate-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94,"normalizedScore":3.6188,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaopennessindex-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50,"normalizedScore":33.4,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-aaifbench-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.5,"normalizedScore":90.5605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-hmmtfeb2026-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":95.2,"normalizedScore":97.3367,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-imoanswerbench-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":89.8,"normalizedScore":99.6344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-apex-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":38.3,"normalizedScore":85.941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-max-apexshortlist-2026-08-01","modelSlug":"deepseek-v4-pro-max","modelName":"DeepSeek V4 Pro (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-terminalbench2-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":66.7,"normalizedScore":55.1601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-browsecomp-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.2,"normalizedScore":81.1715,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-osworldverified-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":73.1,"normalizedScore":74.1304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-toolathlon-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":50,"normalizedScore":47.4333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mcpatlas-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":55.9,"normalizedScore":45.0512,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-claweval-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.3,"normalizedScore":79.3296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-deepsearchqa-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":92.5,"normalizedScore":92.236,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-wideresearch-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":80.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aaagenticindex-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.27,"normalizedScore":54.5554,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-tau2bench-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gdpvalaanormalized-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.4,"normalizedScore":50.514,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gdpvalaa-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1188,"normalizedScore":65.9768,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-apexagentsaa-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":28.5,"normalizedScore":59.9138,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gertlabs-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":56.82,"normalizedScore":65.8707,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-researchclawbench-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18,"normalizedScore":64.3678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-osworld2-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.6549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-sweverified-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.2,"normalizedScore":78.1768,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-livecodebenchv6-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":89.6,"normalizedScore":93.9678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-swepro-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":58.6,"normalizedScore":41.9786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-swemultilingual-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":76.7,"normalizedScore":61.4428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-scicode-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":52.2,"normalizedScore":76.1329,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-vibecodebench-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":37.891,"normalizedScore":53.3654,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aacodingindex-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61.77,"normalizedScore":77.0883,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aascicode-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.5,"normalizedScore":88.7015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-lcr-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-critpt-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":8,"normalizedScore":24.7678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mmmupro-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79.4,"normalizedScore":52.9032,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mmmupropython-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":80.1,"normalizedScore":70.1987,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-charxiv-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":80.4,"normalizedScore":67.8922,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mathvision-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.4,"normalizedScore":65.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-vstar-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":96.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aammmupro-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":79.4,"normalizedScore":90.8935,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-designarenawebsite-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1302,"normalizedScore":87.5415,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gpqa-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.5,"normalizedScore":92.8663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-gpqadiamond-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":90.5,"normalizedScore":92.8663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-hle-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":34.7,"normalizedScore":47.3684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aaomniscienceindex-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.4,"normalizedScore":73.4694,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-omniscienceaccuracy-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.8,"normalizedScore":50.8591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-omnisciencehallucinationrate-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.3,"normalizedScore":69.6019,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aaifbench-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76,"normalizedScore":89.823,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-aime2026-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":96.4,"normalizedScore":95.2365,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-hmmtfeb2026-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.7,"normalizedScore":93.8324,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-mmanswerbench-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86,"normalizedScore":58.6777,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-frontiermathv2tiers13-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":38.966,"normalizedScore":43.782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-2-6-frontiermathv2tier4-2026-08-01","modelSlug":"kimi-k2-6","modelName":"Kimi K2.6","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.58,"normalizedScore":17.5663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-terminalbench2-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":63.5,"normalizedScore":49.4662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-browsecomp-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":68,"normalizedScore":49.3724,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-tau3bench-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.6,"normalizedScore":19.3798,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-mcpatlas-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":71.8,"normalizedScore":72.1843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-cybergym-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":68.7,"normalizedScore":58.3524,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-claweval-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.3,"normalizedScore":79.3296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aaagenticindex-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.87,"normalizedScore":53.828,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-tau2bench-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":97.7,"normalizedScore":98.5873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gdpvalaanormalized-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.8,"normalizedScore":55.5066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gertlabs-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":60.11,"normalizedScore":72.8233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gdpvalaa-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1256,"normalizedScore":69.4094,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-researchclawbench-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18.2,"normalizedScore":66.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-swepro-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":58.4,"normalizedScore":41.4439,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-nl2repo-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":42.7,"normalizedScore":57.4074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-swerebench-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":62.7,"normalizedScore":89.0295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-vibecodebench-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":31.456,"normalizedScore":44.3024,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aacodingindex-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.78,"normalizedScore":68.6219,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aascicode-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.8,"normalizedScore":72.344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-lcr-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.3,"normalizedScore":82.2985,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-critpt-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.6,"normalizedScore":14.2415,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-designarenawebsite-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1299,"normalizedScore":87.0432,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-gpqadiamond-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":86.2,"normalizedScore":86.7313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-hle-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":52.3,"normalizedScore":78.2456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aaomniscienceindex-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.9,"normalizedScore":69.9372,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-omniscienceaccuracy-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.2,"normalizedScore":36.0825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-omnisciencehallucinationrate-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.4,"normalizedScore":81.544,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aaifbench-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.3,"normalizedScore":90.2655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-aime2026-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.3,"normalizedScore":93.3651,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-hmmtnov2025-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":94,"normalizedScore":62.8205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-hmmtfeb2026-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":82.6,"normalizedScore":79.6748,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-mmanswerbench-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.8,"normalizedScore":40.4959,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-frontiermathv2tiers13-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":33.448,"normalizedScore":37.582,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-1-frontiermathv2tier4-2026-08-01","modelSlug":"glm-5-1","modelName":"GLM-5.1","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":12.5,"normalizedScore":15.0602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-terminalbench2-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":59.1,"normalizedScore":41.637,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-osworldverified-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":72.1,"normalizedScore":71.9565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-claweval-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":67.8,"normalizedScore":87.0112,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-cybergym-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":65.2,"normalizedScore":50.3432,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-tau2bench-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":79.5,"normalizedScore":80.222,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-gertlabs-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":62.92,"normalizedScore":78.7616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-osworld2-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":8.3,"normalizedScore":8.1121,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-jobbench-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":36.9,"normalizedScore":61.4468,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-sweverified-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":79.6,"normalizedScore":77.3481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-swerebench-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":60.7,"normalizedScore":80.5907,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-reactnativeevals-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":80.6,"normalizedScore":38.247,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-vibecodebench-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":51.476,"normalizedScore":72.4983,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aascicode-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.9,"normalizedScore":77.5717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-frontiercode-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":24.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-lcr-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":57.7,"normalizedScore":76.2219,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-critpt-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-charxiv-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":77.4,"normalizedScore":60.5392,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aammmupro-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":70.6,"normalizedScore":75.7732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-designarenawebsite-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1310,"normalizedScore":88.8704,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-gpqa-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":89.9,"normalizedScore":92.0103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-supergpqa-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-mmlupro-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":79.2,"normalizedScore":85.202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-hle-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":49,"normalizedScore":72.4561,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aagpqadiamond-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":79.9,"normalizedScore":79.8295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aaomniscienceindex-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-2.9,"normalizedScore":66.1695,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-omniscienceaccuracy-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38,"normalizedScore":59.7938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-omnisciencehallucinationrate-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.9,"normalizedScore":37.5151,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-aaifbench-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":41.2,"normalizedScore":38.4956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-frontiermathv2tiers13-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":32.4,"normalizedScore":36.4045,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-6-frontiermathv2tier4-2026-08-01","modelSlug":"claude-sonnet-4-6","modelName":"Claude Sonnet 4.6","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":8.3,"normalizedScore":10,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-terminalbench2-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":61.6,"normalizedScore":46.0854,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-claweval-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":58.8,"normalizedScore":74.4413,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-qwenclawbench-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":57.2,"normalizedScore":43.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-tau3bench-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.7,"normalizedScore":19.7674,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-vitabench-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":44.3,"normalizedScore":88.8889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-deepplanning-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":41.5,"normalizedScore":56.5762,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-toolathlon-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":39.8,"normalizedScore":26.4887,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mcpatlas-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":48.2,"normalizedScore":31.9113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mcptasks-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.1,"normalizedScore":99.3377,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-wideresearch-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.3,"normalizedScore":68.599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aaagenticindex-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.55,"normalizedScore":49.609,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-tau2bench-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":97.7,"normalizedScore":98.5873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gdpvalaanormalized-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.9,"normalizedScore":46.8429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gdpvalaa-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1139,"normalizedScore":63.5033,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gertlabs-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":50.6,"normalizedScore":52.7261,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-researchclawbench-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":18,"normalizedScore":64.3678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-sweverified-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":78.8,"normalizedScore":76.2431,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-swepro-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.6,"normalizedScore":36.631,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-swemultilingual-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.8,"normalizedScore":54.2289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-livecodebenchv6-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":87.1,"normalizedScore":89.7788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-vibecodebench-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":25.564,"normalizedScore":36.0041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aacodingindex-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.53,"normalizedScore":66.8551,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aascicode-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.7,"normalizedScore":67.1164,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aineedle-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":68.3,"normalizedScore":46.729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-longbenchv2-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":62,"normalizedScore":87.8173,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-lcr-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-critpt-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.9,"normalizedScore":8.9783,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmmu-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":86,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmmupro-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.8,"normalizedScore":50.9677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mathvision-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88,"normalizedScore":68.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-videommmu-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84,"normalizedScore":7.6923,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-screenspotpro-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":68.2,"normalizedScore":53.3175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-charxiv-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":81.5,"normalizedScore":70.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-vstar-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":96.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aammmupro-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78,"normalizedScore":88.488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-designarenawebsite-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1265,"normalizedScore":81.3953,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-gpqa-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.4,"normalizedScore":92.7236,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-supergpqa-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":71.6,"normalizedScore":67.4367,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmlupro-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":88.5,"normalizedScore":98.4348,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmluredux-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.5,"normalizedScore":92.0874,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-ceval-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":93.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hle-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":28.8,"normalizedScore":37.0175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aagpqadiamond-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":88.2,"normalizedScore":91.6193,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aaomniscienceindex-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.7,"normalizedScore":70.5651,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-omniscienceaccuracy-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.2,"normalizedScore":39.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-omnisciencehallucinationrate-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32,"normalizedScore":78.4077,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmluprox-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":84.7,"normalizedScore":69.7368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-nova63-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":57.9,"normalizedScore":70,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-ifeval-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.3,"normalizedScore":97.9314,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-ifbench-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":75.8,"normalizedScore":80.2575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aaifbench-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.2,"normalizedScore":88.6431,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-aime2026-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.3,"normalizedScore":93.3651,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hmmtfeb2025-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":96.7,"normalizedScore":88.2353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hmmtnov2025-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":94.6,"normalizedScore":70.5128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-hmmtfeb2026-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.8,"normalizedScore":86.9638,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-mmanswerbench-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.8,"normalizedScore":40.4959,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-frontiermathv2tiers13-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":26.207,"normalizedScore":29.4461,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-plus-frontiermathv2tier4-2026-08-01","modelSlug":"qwen3-6-plus","modelName":"Qwen3.6 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":8.333,"normalizedScore":10.0398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-terminalbench2-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":56.2,"normalizedScore":36.4769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-claweval-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.7,"normalizedScore":72.905,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-qwenclawbench-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":54.1,"normalizedScore":18.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-tau3bench-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":65.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-deepplanning-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":14.6,"normalizedScore":0.4175,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-toolathlon-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":38,"normalizedScore":22.7926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mcpatlas-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":31.1,"normalizedScore":2.7304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mcptasks-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":60.8,"normalizedScore":11.2583,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-wideresearch-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":69.8,"normalizedScore":46.8599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-tau2bench-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.2,"normalizedScore":99.0918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-cybergym-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":43.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-apexagentsaa-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":14.5,"normalizedScore":29.7414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-gertlabs-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":50.99,"normalizedScore":53.5503,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-sweverified-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.8,"normalizedScore":74.8619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-sweverifiedarcee-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":72.8,"normalizedScore":77.4194,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-swepro-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.1,"normalizedScore":32.6203,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-swemultilingual-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.3,"normalizedScore":52.9851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-swerebench-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":62.8,"normalizedScore":89.4515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-reactnativeevals-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":74.8,"normalizedScore":15.1394,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aascicode-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.2,"normalizedScore":76.3912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-longbenchv2-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.8,"normalizedScore":81.7259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aineedle-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":63.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-lcr-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.3,"normalizedScore":83.6196,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-critpt-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2,"normalizedScore":6.192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-designarenawebsite-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1274,"normalizedScore":82.8904,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-gpqa-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":86,"normalizedScore":86.446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-gpqadiamond-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":86,"normalizedScore":86.446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-supergpqa-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":66.8,"normalizedScore":60.757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmlupro-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.7,"normalizedScore":94.4508,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmluproarcee-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":85.8,"normalizedScore":76.259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hle-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":50.4,"normalizedScore":74.9123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aaomniscienceindex-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2,"normalizedScore":70.0157,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-omniscienceaccuracy-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.9,"normalizedScore":40.7216,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-omnisciencehallucinationrate-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34,"normalizedScore":75.9952,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmluprox-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":83.1,"normalizedScore":48.6842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-nova63-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":55.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-ifeval-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":92.6,"normalizedScore":92.9078,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aaifbench-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.3,"normalizedScore":84.3658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aime2026-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.8,"normalizedScore":94.2157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-aime2025arcee-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":93.3,"normalizedScore":91.4248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hmmtfeb2025-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":97.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hmmtnov2025-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":96.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-hmmtfeb2026-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.4,"normalizedScore":85.0014,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-mmanswerbench-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":82.5,"normalizedScore":29.7521,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-frontiermathv2tiers13-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":16.434,"normalizedScore":18.4652,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-frontiermathv2tier4-2026-08-01","modelSlug":"glm-5","modelName":"GLM-5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.1,"normalizedScore":2.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-terminalbench2-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":64.7,"normalizedScore":51.6014,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-browsecomp-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":77.4,"normalizedScore":69.0377,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-mcpatlas-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":79.6,"normalizedScore":85.4949,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-toolathlonverified-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-toolathlonverified","benchmarkName":"Toolathlon-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":54.4,"normalizedScore":15.2104,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-sweverified-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.2,"normalizedScore":78.1768,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-swepro-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.9,"normalizedScore":34.7594,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-scicode-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":48.7,"normalizedScore":65.5589,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-arcagi2-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":40.1,"normalizedScore":33.5868,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-critpt-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":8.3,"normalizedScore":25.6966,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-mmmupro-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":74,"normalizedScore":35.4839,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-charxiv-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":81.3,"normalizedScore":70.098,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-charxivnotools-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":77.4,"normalizedScore":3.3613,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-gpqa-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":89.5,"normalizedScore":91.4396,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-gpqadiamond-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":89.5,"normalizedScore":91.4396,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-hle-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":47.8,"normalizedScore":70.3509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-hlenotools-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":31.6,"normalizedScore":49.0706,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-ifbench-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":82.2,"normalizedScore":93.9914,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-aime2026-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.5,"normalizedScore":93.7053,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-small-hmmtfeb2026-2026-08-01","modelSlug":"inkling-small","modelName":"Inkling-Small","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":90.2,"normalizedScore":90.328,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling-Small; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-terminalbench2-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":63.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-browsecomp-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":77.1,"normalizedScore":68.41,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-mcpatlas-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":74.1,"normalizedScore":76.1092,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-designarenaagenticwebdev-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-designarenaagenticwebdev","benchmarkName":"Design Arena Agentic Web Dev Elo","benchmarkCategory":"agents","benchmarkOrganisation":"Design Arena / Intelligence","benchmarkVersion":"2026","score":1257,"normalizedScore":50,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aaagenticindex-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.34,"normalizedScore":58.3197,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gdpvalaanormalized-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":36.8,"normalizedScore":54.0382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gdpvalaa-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1237,"normalizedScore":68.4503,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aabriefcaseelo-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":839,"normalizedScore":26.8272,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aatau3banking-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.7,"normalizedScore":48.9474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aaenterpriseopsgym-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.1,"normalizedScore":49.2188,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aaterminalbench21-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-sweverified-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.6,"normalizedScore":74.5856,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-swepro-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":54.3,"normalizedScore":30.4813,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aacodingindex-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.06,"normalizedScore":63.364,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aascicode-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.1,"normalizedScore":76.2226,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-lcr-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.3,"normalizedScore":83.6196,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-critpt-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.4,"normalizedScore":16.7183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-mmmupro-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":73.5,"normalizedScore":33.871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-charxiv-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":82,"normalizedScore":71.8137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-charxivnotools-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-charxivnotools","benchmarkName":"CharXiv Reasoning without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv authors","benchmarkVersion":"2024","score":78.1,"normalizedScore":9.2437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aammmupro-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":73.5,"normalizedScore":80.756,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gpqa-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.9,"normalizedScore":89.1568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-gpqadiamond-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87.9,"normalizedScore":89.1568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-hle-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":46,"normalizedScore":67.193,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-hlenotools-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":30,"normalizedScore":46.0967,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aaomniscienceindex-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.1,"normalizedScore":70.0942,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-omniscienceaccuracy-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40,"normalizedScore":63.2302,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-omnisciencehallucinationrate-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.1,"normalizedScore":40.8926,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aaopennessindex-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-ifbench-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":79.8,"normalizedScore":88.8412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-inkling-aime2026-2026-08-01","modelSlug":"inkling","modelName":"Inkling","providerId":"thinking-machines-lab","providerName":"Thinking Machines Lab","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":97.1,"normalizedScore":96.4274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Inkling; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-terminalbench2-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":66,"normalizedScore":53.9146,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-browsecomp-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":83.52,"normalizedScore":81.841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-osworldverified-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":70.06,"normalizedScore":67.5217,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-mcpatlas-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":74.2,"normalizedScore":76.2799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-claweval-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":74.5,"normalizedScore":96.3687,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaagenticindex-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.36,"normalizedScore":63.8116,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-tau2bench-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":88.9,"normalizedScore":89.7074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-gdpvalaanormalized-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.5,"normalizedScore":65.3451,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-gdpvalaa-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1391,"normalizedScore":76.2241,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-gdpvalrubrics-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gdpvalrubrics","benchmarkName":"GDPval rubrics","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":74.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-bankertoolbench-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-bankertoolbench","benchmarkName":"BankerToolBench","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":76.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-researchclawbench-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":19.8,"normalizedScore":85.0575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-osworld2-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.6549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aabriefcaseelo-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1110,"normalizedScore":49.3355,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaenterpriseopsgym-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.1,"normalizedScore":25.7813,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaharveylab-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.4,"normalizedScore":92.3172,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-terminalbenchhard-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":42.4,"normalizedScore":44.5755,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaterminalbench21-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.2,"normalizedScore":29.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-sweverified-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.5,"normalizedScore":78.5912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-swepro-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":59,"normalizedScore":43.0481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-nl2repo-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":42.13,"normalizedScore":55.2963,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aacodingindex-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.57,"normalizedScore":72.5654,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aascicode-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.4,"normalizedScore":75.0422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-vibev2-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibev2","benchmarkName":"VIBE V2","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":50.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-svgbench-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-svgbench","benchmarkName":"SVG-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":63.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-kernelbenchhard-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-kernelbenchhard","benchmarkName":"KernelBench Hard","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":28.8,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-lcr-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-critpt-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.7,"normalizedScore":11.4551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-officeqapro-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"officeqa-pro","benchmarkName":"OfficeQA Pro","benchmarkCategory":"agents","benchmarkOrganisation":"OfficeQA authors","benchmarkVersion":"2026","score":45.1,"normalizedScore":6.4378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-omnidocbench15-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omnidocbench15","benchmarkName":"OmniDocBench 1.5","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":91.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-mmmupro-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.1,"normalizedScore":48.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-videommmu-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.6,"normalizedScore":23.0769,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-videommewithsub-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-designarenawebsite-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1286,"normalizedScore":84.8837,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aammmupro-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.6,"normalizedScore":89.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aagpqadiamond-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":92.9,"normalizedScore":98.2955,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aahle-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":37.1,"normalizedScore":67.7291,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaomniscienceindex-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.4,"normalizedScore":69.5447,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-omniscienceaccuracy-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15,"normalizedScore":20.2749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-omnisciencehallucinationrate-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.1,"normalizedScore":97.5875,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaopennessindex-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-aaifbench-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":82.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m3-usamo2026-2026-08-01","modelSlug":"minimax-m3","modelName":"MiniMax M3","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"usamo-2026","benchmarkName":"USAMO 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":85.71,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-terminalbench2-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":59.3,"normalizedScore":41.9929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-osworldverified-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":66.3,"normalizedScore":59.3478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-osworld-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":"2026","score":66.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-claweval-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":59.6,"normalizedScore":75.5587,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-qwenclawbench-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":52.3,"normalizedScore":4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-tau3bench-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.2,"normalizedScore":17.8295,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-vitabench-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":23.3,"normalizedScore":24.0741,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-deepplanning-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":26.4,"normalizedScore":25.0522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-toolathlon-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":43.5,"normalizedScore":34.0862,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mcpatlas-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":42.3,"normalizedScore":21.843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mcptasks-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":71.8,"normalizedScore":84.106,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-wideresearch-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":76.4,"normalizedScore":78.744,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-cybergym-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":50.6,"normalizedScore":16.9336,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-tau2bench-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86.3,"normalizedScore":87.0838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-gertlabs-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":64.23,"normalizedScore":81.53,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-jobbench-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":32.3,"normalizedScore":51.4836,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-sweverified-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80.9,"normalizedScore":79.1436,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-livecodebenchv6-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":84.8,"normalizedScore":85.9249,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-swepro-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.1,"normalizedScore":37.9679,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-swemultilingual-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":77.5,"normalizedScore":63.4328,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-nl2repo-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":43.2,"normalizedScore":59.2593,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aascicode-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47,"normalizedScore":77.7403,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-longbenchv2-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":64.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aineedle-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-lcr-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.3,"normalizedScore":86.2616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-critpt-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmmupro-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":70.6,"normalizedScore":24.5161,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mathvision-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-charxiv-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":68.5,"normalizedScore":38.7255,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-videommmu-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.4,"normalizedScore":17.9487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-screenspotpro-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":45.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-vstar-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":67,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aammmupro-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":71.2,"normalizedScore":76.8041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-designarenawebsite-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1272,"normalizedScore":82.5581,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-gpqa-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-supergpqa-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":70.6,"normalizedScore":66.0451,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmlupro-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":89.5,"normalizedScore":99.8577,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmluredux-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":96.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-ceval-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":92.2,"normalizedScore":66.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hle-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":30.8,"normalizedScore":40.5263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aagpqadiamond-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81,"normalizedScore":81.392,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aaomniscienceindex-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-3.9,"normalizedScore":65.3846,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-omniscienceaccuracy-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.7,"normalizedScore":64.433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-omnisciencehallucinationrate-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.4,"normalizedScore":26.0555,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aammlupro-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":88.9,"normalizedScore":90,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmluprox-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":85.7,"normalizedScore":82.8947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-nova63-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":56.7,"normalizedScore":40,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-ifeval-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":90.9,"normalizedScore":87.8842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-ifbench-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":58,"normalizedScore":42.0601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aaifbench-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43,"normalizedScore":41.1504,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-aime2026-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.1,"normalizedScore":93.0248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hmmtfeb2025-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":92.9,"normalizedScore":32.3529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hmmtnov2025-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":93.3,"normalizedScore":53.8462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-hmmtfeb2026-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.3,"normalizedScore":83.4595,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-mmanswerbench-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84,"normalizedScore":42.1488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-frontiermathv2tiers13-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":20.69,"normalizedScore":23.2472,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-frontiermathv2tier4-2026-08-01","modelSlug":"claude-opus-4-5","modelName":"Claude Opus 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-terminalbench2-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":52.5,"normalizedScore":29.8932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-browsecomp-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":62,"normalizedScore":36.8201,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-claweval-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":56.8,"normalizedScore":71.648,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-qwenclawbench-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":51.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-tau3bench-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":68.4,"normalizedScore":10.8527,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-vitabench-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":43.7,"normalizedScore":87.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-deepplanning-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":37.6,"normalizedScore":48.4342,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-toolathlon-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":36.3,"normalizedScore":19.3018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mcpatlas-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":46.1,"normalizedScore":28.3276,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mcptasks-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-wideresearch-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":74,"normalizedScore":67.1498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-tau2bench-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.6,"normalizedScore":96.4682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gertlabs-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":46.76,"normalizedScore":44.6112,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-researchclawbench-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":14.2,"normalizedScore":20.6897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aaagenticindex-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.85,"normalizedScore":35.6065,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-apexagentsaa-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":15.3,"normalizedScore":31.4655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gdpvalaanormalized-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.1,"normalizedScore":33.9207,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gdpvalaa-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":962,"normalizedScore":54.5684,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-sweverified-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":76.2,"normalizedScore":72.6519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-livecodebenchv6-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":83.6,"normalizedScore":83.9142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-swepro-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":50.9,"normalizedScore":21.3904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aascicode-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42,"normalizedScore":69.3086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aacodingindex-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.21,"normalizedScore":57.9223,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-longbenchv2-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":63.2,"normalizedScore":93.9086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aineedle-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aineedle","benchmarkName":"AI-Needle","benchmarkCategory":"reasoning","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":68.7,"normalizedScore":50.4673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-lcr-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.7,"normalizedScore":86.79,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-critpt-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.7,"normalizedScore":5.2632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmmupro-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79,"normalizedScore":51.6129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mathvision-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88.6,"normalizedScore":71.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-charxiv-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":80.8,"normalizedScore":68.8725,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-videommmu-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.7,"normalizedScore":25.641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-screenspotpro-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":65.6,"normalizedScore":47.1564,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-vstar-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":95.8,"normalizedScore":96.3211,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aammmupro-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":77.3,"normalizedScore":87.2852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-gpqa-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":88.4,"normalizedScore":89.8702,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-supergpqa-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":70.4,"normalizedScore":65.7668,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmlupro-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":87.8,"normalizedScore":97.4388,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmluredux-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.9,"normalizedScore":93.5946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-ceval-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":93,"normalizedScore":90.9091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hle-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":28.7,"normalizedScore":36.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aagpqadiamond-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.3,"normalizedScore":93.1818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aaomniscienceindex-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-29.8,"normalizedScore":45.0549,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-omniscienceaccuracy-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.4,"normalizedScore":48.4536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-omnisciencehallucinationrate-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.1,"normalizedScore":9.5296,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmluprox-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":84.7,"normalizedScore":69.7368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-nova63-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-ifeval-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":92.6,"normalizedScore":92.9078,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aaifbench-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":78.8,"normalizedScore":93.9528,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-aime2026-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":93.3,"normalizedScore":89.9626,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hmmtfeb2025-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":94.8,"normalizedScore":60.2941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hmmtnov2025-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":92.7,"normalizedScore":46.1538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-hmmtfeb2026-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.9,"normalizedScore":87.104,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-397b-mmanswerbench-2026-08-01","modelSlug":"qwen3-5-397b","modelName":"Qwen3.5 397B A17B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":80.9,"normalizedScore":16.5289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-terminalbench2-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":60,"normalizedScore":43.2384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-osworldverified-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":72.1,"normalizedScore":71.9565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-mcpatlas-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":57.7,"normalizedScore":48.1229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-toolathlon-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":42.9,"normalizedScore":32.8542,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-tau2bench-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.4,"normalizedScore":94.2482,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aaagenticindex-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.17,"normalizedScore":54.3735,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-apexagentsaa-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":28.2,"normalizedScore":59.2672,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.5,"normalizedScore":49.1924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-gdpvalaa-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1169,"normalizedScore":65.0177,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-vibecodebench-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":47.969,"normalizedScore":67.5591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aacodingindex-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.08,"normalizedScore":69.0459,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aascicode-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49.9,"normalizedScore":82.6307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-frontiercode-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":27,"normalizedScore":9.2466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-lcr-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":91.5456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-critpt-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":10,"normalizedScore":30.9598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-mmmupro-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":76.6,"normalizedScore":43.871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-mmmupropython-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":78,"normalizedScore":56.2914,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aammmupro-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":73.3,"normalizedScore":80.4124,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-gpqa-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":88,"normalizedScore":89.2995,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-hle-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":41.5,"normalizedScore":59.2982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-hlenotools-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":28.2,"normalizedScore":42.7509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aagpqadiamond-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87.5,"normalizedScore":90.625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-18.7,"normalizedScore":53.7677,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.5,"normalizedScore":58.9347,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.8,"normalizedScore":8.6852,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-aaifbench-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.3,"normalizedScore":85.8407,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":28.28,"normalizedScore":31.7753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-mini-frontiermathv2tier4-2026-08-01","modelSlug":"gpt-5-4-mini","modelName":"GPT-5.4 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.08,"normalizedScore":2.506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-pro-tau2bench-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":87.1,"normalizedScore":87.891,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-gertlabs-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":63.23,"normalizedScore":79.4167,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-jobbench-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":11.4,"normalizedScore":6.2162,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-vibecodebench-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":14.3,"normalizedScore":20.14,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aascicode-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":93.086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aalivecodebench-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aalivecodebench","benchmarkName":"Artificial Analysis LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-arcagi2-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":31.1,"normalizedScore":22.18,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-lcr-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":70.7,"normalizedScore":93.395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-critpt-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":9.1,"normalizedScore":28.1734,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-mmmupro-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":81,"normalizedScore":58.0645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-mathvision-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.6,"normalizedScore":61.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-videommmu-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":87.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-screenspotpro-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":72.7,"normalizedScore":63.981,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-charxiv-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":81.4,"normalizedScore":70.3431,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-vstar-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":88,"normalizedScore":70.2341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aammmupro-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.2,"normalizedScore":92.268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aagpqadiamond-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":90.8,"normalizedScore":95.3125,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aahle-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":37.2,"normalizedScore":67.9283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aaomniscienceindex-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.8,"normalizedScore":80.8477,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-omniscienceaccuracy-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.9,"normalizedScore":90.5498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.9,"normalizedScore":7.3583,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aammlupro-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aaglobalmmlulite-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":92.2,"normalizedScore":90.3846,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-aaifbench-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.4,"normalizedScore":81.5634,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-frontiermathv2tiers13-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":37.6,"normalizedScore":42.2472,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-frontiermathv2tier4-2026-08-01","modelSlug":"gemini-3-pro","modelName":"Gemini 3 Pro Preview","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":18.75,"normalizedScore":22.5904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-kimi-k2-5-terminalbench2-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":50.8,"normalizedScore":26.8683,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-browsecomp-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":60.6,"normalizedScore":33.8912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-claweval-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":52.3,"normalizedScore":65.3631,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-qwenclawbench-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":54.3,"normalizedScore":20,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-tau3bench-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":65.7,"normalizedScore":0.3876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-deepsearchqa-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":77.1,"normalizedScore":44.4099,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-deepplanning-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":14.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-toolathlon-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":27.8,"normalizedScore":1.848,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mcpatlas-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":29.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mcptasks-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mcptasks","benchmarkName":"MCP-Tasks","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-wideresearch-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":72.7,"normalizedScore":60.8696,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-tau2bench-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-apexagentsaa-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":11.5,"normalizedScore":23.2759,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gertlabs-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":45.88,"normalizedScore":42.7515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-researchclawbench-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":14,"normalizedScore":18.3908,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-jobbench-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":8.73,"normalizedScore":0.4332,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aaagenticindex-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.69,"normalizedScore":38.9525,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gdpvalaanormalized-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.1,"normalizedScore":36.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gdpvalaa-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1003,"normalizedScore":56.6381,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-sweverified-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":76.8,"normalizedScore":73.4807,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-sweverifiedarcee-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":70.8,"normalizedScore":61.2903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-livecodebenchv6-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":85,"normalizedScore":86.2601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-swepro-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":50.7,"normalizedScore":20.8556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-swemultilingual-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73,"normalizedScore":52.2388,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-swerebench-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.5,"normalizedScore":71.308,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-reactnativeevals-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":77.2,"normalizedScore":24.7012,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-scicode-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":48.7,"normalizedScore":65.5589,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aascicode-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49,"normalizedScore":81.113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aacodingindex-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.78,"normalizedScore":55.9011,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-longbenchv2-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":61,"normalizedScore":82.7411,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-lcr-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.3,"normalizedScore":86.2616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-critpt-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.1,"normalizedScore":9.5975,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmmupro-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-videomme-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-videomme","benchmarkName":"Video-MME","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MME benchmark team","benchmarkVersion":"2024","score":87.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmvu-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":80.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-videommmu-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":86.6,"normalizedScore":74.359,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aammmupro-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.4,"normalizedScore":84.0206,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-designarenawebsite-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1275,"normalizedScore":83.0565,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gpqa-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.6,"normalizedScore":88.7288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-gpqadiamond-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87.6,"normalizedScore":88.7288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-supergpqa-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":69.2,"normalizedScore":64.0969,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmlupro-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":87.1,"normalizedScore":96.4428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmluproarcee-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":87.1,"normalizedScore":85.6115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hle-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":30.1,"normalizedScore":39.2982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aaomniscienceindex-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-8.1,"normalizedScore":62.0879,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-omniscienceaccuracy-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.3,"normalizedScore":53.4364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-omnisciencehallucinationrate-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.6,"normalizedScore":39.0832,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmluprox-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":82.3,"normalizedScore":38.1579,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-nova63-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-nova63","benchmarkName":"NOVA-63","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":56,"normalizedScore":22.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-ifeval-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":93.9,"normalizedScore":96.7494,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aaifbench-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.2,"normalizedScore":81.2684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aime2025-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":96.1,"normalizedScore":98.4093,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aime2026-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":95.8,"normalizedScore":94.2157,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-aime2025arcee-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":96.3,"normalizedScore":95.3826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hmmtfeb2025-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":95.4,"normalizedScore":69.1176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hmmtnov2025-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":91.1,"normalizedScore":25.641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-hmmtfeb2026-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.1,"normalizedScore":85.9826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-mmanswerbench-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.8,"normalizedScore":23.9669,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-frontiermathv2tiers13-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":27.9,"normalizedScore":31.3483,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-5-frontiermathv2tier4-2026-08-01","modelSlug":"kimi-k2-5","modelName":"Kimi K2.5","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.2,"normalizedScore":5.0602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-terminalbench2-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":56.4,"normalizedScore":36.8327,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-pinchbench-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-pinchbench","benchmarkName":"PinchBench","benchmarkCategory":"agents","benchmarkOrganisation":"Kilo Code","benchmarkVersion":"2026","score":90,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-browsecomp-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":44.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-tau3bench-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":70.9,"normalizedScore":20.5426,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gdpvalaanormalized-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.1,"normalizedScore":48.605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-hlewithtools-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":37.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaagenticindex-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.36,"normalizedScore":49.2635,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-tau2bench-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83.3,"normalizedScore":84.0565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gdpvalaa-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1162,"normalizedScore":64.6643,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aabriefcaseelo-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":873,"normalizedScore":29.6512,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaenterpriseopsgym-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.9,"normalizedScore":13.2812,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaharveylab-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.7,"normalizedScore":84.0149,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-terminalbenchhard-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":36.4,"normalizedScore":30.4245,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-sweverified-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":71.9,"normalizedScore":66.7127,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-swemultilingual-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":67.7,"normalizedScore":39.0547,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-livecodebenchv6-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":89,"normalizedScore":92.9625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-scicode-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":44.6,"normalizedScore":53.1722,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aacodingindex-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.27,"normalizedScore":59.4205,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aascicode-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.9,"normalizedScore":65.7673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-lcr-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67,"normalizedScore":88.5073,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-critpt-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.1,"normalizedScore":9.5975,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-longbenchv2-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":61.9,"normalizedScore":87.3096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-designarenawebsite-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1127,"normalizedScore":58.4718,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gpqa-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-gpqadiamond-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-hle-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":26.7,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-hlenotools-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":26.7,"normalizedScore":39.9628,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-mmlupro-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.8,"normalizedScore":96.0159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-omniscienceaccuracy-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.6,"normalizedScore":31.6151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaomniscienceindex-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-0.8,"normalizedScore":67.8179,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-omnisciencehallucinationrate-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.5,"normalizedScore":82.6297,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaopennessindex-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.3,"normalizedScore":100,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-mmluprox-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":83,"normalizedScore":47.3684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-ifbench-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":81.7,"normalizedScore":92.9185,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-ultra-aaifbench-2026-08-01","modelSlug":"nemotron-3-ultra","modelName":"Nemotron 3 Ultra","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":81.4,"normalizedScore":97.7876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-terminalbench2-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":49.4,"normalizedScore":24.3772,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-browsecomp-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":63.8,"normalizedScore":40.5858,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-osworldverified-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":58,"normalizedScore":41.3043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-tau2bench-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.6,"normalizedScore":94.4501,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aaagenticindex-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.72,"normalizedScore":37.1886,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-gdpvalaanormalized-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.1,"normalizedScore":35.3891,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-gdpvalaa-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":982,"normalizedScore":55.578,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-sweverified-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":72,"normalizedScore":66.8508,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aacodingindex-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.71,"normalizedScore":54.3887,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aascicode-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42,"normalizedScore":69.3086,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-longbenchv2-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.2,"normalizedScore":78.6802,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-lcr-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-critpt-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmmu-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":83.9,"normalizedScore":96.0623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmvu-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":74.7,"normalizedScore":29.6296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mathvision-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":59.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-charxiv-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":77.2,"normalizedScore":60.049,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-vstar-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":93.2,"normalizedScore":87.6254,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aammmupro-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75,"normalizedScore":83.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmlupro-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.7,"normalizedScore":95.8736,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-supergpqa-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":67.1,"normalizedScore":61.1745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-gpqa-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":86.6,"normalizedScore":87.302,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aagpqadiamond-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.7,"normalizedScore":88.0682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aahle-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.4,"normalizedScore":40.4382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aaomniscienceindex-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-39.6,"normalizedScore":37.3626,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-omniscienceaccuracy-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.7,"normalizedScore":36.9416,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-omnisciencehallucinationrate-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":85.5,"normalizedScore":13.8721,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-mmluprox-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":82.2,"normalizedScore":36.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-ifeval-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":93.4,"normalizedScore":95.2719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-122b-a10b-aaifbench-2026-08-01","modelSlug":"qwen3-5-122b-a10b","modelName":"Qwen3.5-122B-A10B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.7,"normalizedScore":89.3805,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-terminalbench2-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":41.6,"normalizedScore":10.4982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-browsecomp-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":61,"normalizedScore":34.728,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-osworldverified-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":56.2,"normalizedScore":37.3913,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-tau2bench-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.9,"normalizedScore":94.7528,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-gertlabs-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.41,"normalizedScore":29.0786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-sweverified-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":72.4,"normalizedScore":67.4033,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-swerebench-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.9,"normalizedScore":72.9958,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aascicode-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.5,"normalizedScore":65.0927,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-longbenchv2-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.6,"normalizedScore":80.7107,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-lcr-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.3,"normalizedScore":88.9036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-critpt-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmmu-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":82.3,"normalizedScore":93.0621,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmvu-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":73.3,"normalizedScore":12.3457,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mathvision-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86,"normalizedScore":58.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-vstar-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":93.7,"normalizedScore":89.2977,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aammmupro-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75,"normalizedScore":83.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmlupro-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.1,"normalizedScore":95.0199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-supergpqa-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":65.6,"normalizedScore":59.0871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-gpqa-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":85.5,"normalizedScore":85.7326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aagpqadiamond-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.8,"normalizedScore":88.2102,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aahle-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":22.2,"normalizedScore":38.0478,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aaomniscienceindex-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-42,"normalizedScore":35.4788,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-omniscienceaccuracy-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21,"normalizedScore":30.5842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-omnisciencehallucinationrate-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":79.7,"normalizedScore":20.8685,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-mmluprox-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":82.2,"normalizedScore":36.8421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-ifeval-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":95,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-27b-aaifbench-2026-08-01","modelSlug":"qwen3-5-27b","modelName":"Qwen3.5 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.6,"normalizedScore":89.233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-browsecomp-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":65.8,"normalizedScore":44.7699,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-osworldverified-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":47.3,"normalizedScore":18.0435,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-tau2bench-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-gertlabs-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":46.54,"normalizedScore":44.1462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-jobbench-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":34.3,"normalizedScore":55.8155,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-sweverified-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":80,"normalizedScore":77.9006,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-swepro-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":55.6,"normalizedScore":33.9572,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-vibecodebench-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":53.499,"normalizedScore":75.3475,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aascicode-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":52.1,"normalizedScore":86.3406,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-arcagi2-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":52.9,"normalizedScore":49.8099,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-lcr-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.7,"normalizedScore":96.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-critpt-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":11.6,"normalizedScore":35.9133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-mmmupro-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":79.5,"normalizedScore":53.2258,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-mathvision-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83,"normalizedScore":43.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-charxiv-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":82.1,"normalizedScore":72.0588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-vstar-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":75.9,"normalizedScore":29.7659,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-designarenawebsite-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1219,"normalizedScore":73.7542,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-gpqa-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":92.4,"normalizedScore":95.5771,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aagpqadiamond-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":90.3,"normalizedScore":94.6023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aahle-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":35.4,"normalizedScore":64.3426,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-1,"normalizedScore":67.6609,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.8,"normalizedScore":69.7595,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":79.7,"normalizedScore":20.8685,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aaifbench-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.4,"normalizedScore":88.9381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-aaaime2025-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaaime2025","benchmarkName":"Artificial Analysis AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":40.7,"normalizedScore":45.7303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-frontiermathv2tier4-2026-08-01","modelSlug":"gpt-5-2","modelName":"GPT-5.2","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":18.8,"normalizedScore":22.6506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-terminalbench2-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":59.3,"normalizedScore":41.9929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-claweval-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":72.4,"normalizedScore":93.4358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-qwenclawbench-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":53.4,"normalizedScore":12.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-qwenwebbench-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1487,"normalizedScore":52.6316,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-androidworld-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":70.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aaagenticindex-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.03,"normalizedScore":48.6634,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-tau2bench-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.2,"normalizedScore":95.0555,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gdpvalaanormalized-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.9,"normalizedScore":46.8429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gdpvalaa-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1138,"normalizedScore":63.4528,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gertlabs-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":54.84,"normalizedScore":61.6864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-sweverified-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.2,"normalizedScore":74.0331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-swemultilingual-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":71.3,"normalizedScore":48.01,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-swepro-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":53.5,"normalizedScore":28.3422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-livecodebench-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":83.9,"normalizedScore":85.7407,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-nl2repo-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":36.2,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aacodingindex-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":53.72,"normalizedScore":65.7102,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aascicode-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.8,"normalizedScore":65.5987,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-lcr-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.7,"normalizedScore":90.753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-critpt-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmmu-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":82.9,"normalizedScore":94.1871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmmupro-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":75.8,"normalizedScore":41.2903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-realworldqa-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84.1,"normalizedScore":90.1651,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-dynamath-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-dynamath","benchmarkName":"DynaMath","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mstar-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mstar","benchmarkName":"MStar","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.4,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-simplevqa-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":56.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-charxiv-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":78.4,"normalizedScore":62.9902,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-ccocr-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ccocr","benchmarkName":"CC-OCR","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-countbench-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-countbench","benchmarkName":"CountBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":97.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-refcocoavg-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":92.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-erqa-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":62.5,"normalizedScore":59.8901,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-videommewithsub-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.7,"normalizedScore":88.4615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-videommmu-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":84.4,"normalizedScore":17.9487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mlvuavg-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mlvuavg","benchmarkName":"MLVU mean average","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.6,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-vstar-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":94.7,"normalizedScore":92.6421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aammmupro-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.6,"normalizedScore":82.646,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmlupro-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.2,"normalizedScore":95.1622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmluredux-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":93.5,"normalizedScore":88.3195,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-supergpqa-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":66,"normalizedScore":59.6438,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-ceval-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":91.4,"normalizedScore":42.4242,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-gpqa-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.8,"normalizedScore":89.0141,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hle-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":24,"normalizedScore":28.5965,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aagpqadiamond-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.2,"normalizedScore":85.9375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aaomniscienceindex-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-19.8,"normalizedScore":52.9042,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-omniscienceaccuracy-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.2,"normalizedScore":27.4914,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-omnisciencehallucinationrate-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.3,"normalizedScore":58.7455,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aaifbench-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":67.6,"normalizedScore":77.4336,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hmmtfeb2025-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":93.8,"normalizedScore":45.5882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hmmtnov2025-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":90.7,"normalizedScore":20.5128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-hmmtfeb2026-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84.3,"normalizedScore":82.0578,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-mmanswerbench-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":80.8,"normalizedScore":15.7025,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-27b-aime2026-2026-08-01","modelSlug":"qwen3-6-27b","modelName":"Qwen3.6 27B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":94.1,"normalizedScore":91.3236,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-terminalbench2-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":46,"normalizedScore":18.3274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-livecodebenchv6-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":87.7,"normalizedScore":90.7842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-sweverified-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.5,"normalizedScore":68.9227,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-swepro-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.8,"normalizedScore":26.4706,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-graphwalksbfs128k-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-graphwalksbfs128k","benchmarkName":"Graphwalks BFS 0K-128K","benchmarkCategory":"reasoning","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-gpqa-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":84.2,"normalizedScore":83.8779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-gpqadiamond-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":84.2,"normalizedScore":83.8779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-mmlupro-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85,"normalizedScore":93.4548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-simpleqa-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":31,"normalizedScore":22.7011,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-ifbench-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":85,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-aime2025-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":97,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-aime2026-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":94.5,"normalizedScore":92.0041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mai-thinking-1-hmmtfeb2026-2026-08-01","modelSlug":"mai-thinking-1","modelName":"MAI-Thinking-1","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":84.9,"normalizedScore":82.8988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MAI-Thinking-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-terminalbench2-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":59.1,"normalizedScore":41.637,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-mcpatlas-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":69.4,"normalizedScore":68.0887,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-toolathlon-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":46.3,"normalizedScore":39.8357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-claweval-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":59.8,"normalizedScore":75.838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-gertlabs-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":50.28,"normalizedScore":52.0499,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-researchclawbench-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":17.1,"normalizedScore":54.023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-livecodebenchpass1cot-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-livecodebenchpass1cot","benchmarkName":"LiveCodeBench Pass@1 with Chain-of-Thought","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek","benchmarkVersion":"2026","score":56.8,"normalizedScore":4.1775,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-sweverified-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.6,"normalizedScore":69.0608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-swepro-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.1,"normalizedScore":24.5989,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-swemultilingual-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":69.8,"normalizedScore":44.2786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-mrcr1m-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":44.7,"normalizedScore":31.8102,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-corpusqa1m-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":35.6,"normalizedScore":43.2258,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-designarenawebsite-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1260,"normalizedScore":80.5648,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-mmlupro-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":82.9,"normalizedScore":90.4667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-simpleqa-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":45,"normalizedScore":62.931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-chinesesimpleqa-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":75.8,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-gpqa-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":72.9,"normalizedScore":67.7557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-gpqadiamond-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":72.9,"normalizedScore":67.7557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-hle-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":7.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-hmmtfeb2026-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":31.7,"normalizedScore":8.3263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-imoanswerbench-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":35.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-apex-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":0.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-apexshortlist-2026-08-01","modelSlug":"deepseek-v4-pro","modelName":"DeepSeek V4 Pro","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":9.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-terminalbench2-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":40.5,"normalizedScore":8.5409,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-browsecomp-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":61,"normalizedScore":34.728,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-osworldverified-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":54.5,"normalizedScore":33.6957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-tau2bench-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":89.2,"normalizedScore":90.0101,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-gertlabs-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":28.96,"normalizedScore":6.9949,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-sweverified-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":69.2,"normalizedScore":62.9834,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-swerebench-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":53.7,"normalizedScore":51.0549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aascicode-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.7,"normalizedScore":62.0573,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-longbenchv2-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":59,"normalizedScore":72.5888,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-lcr-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.7,"normalizedScore":82.8269,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-critpt-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmmu-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":81.4,"normalizedScore":91.3745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmvu-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmvu","benchmarkName":"Multimodal Multi-disciplinary Video Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMVU benchmark maintainers","benchmarkVersion":"2026","score":72.3,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mathvision-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.9,"normalizedScore":48,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-vstar-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":92.7,"normalizedScore":85.9532,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aammmupro-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.7,"normalizedScore":79.3814,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmlupro-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.3,"normalizedScore":93.8816,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-supergpqa-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":63.4,"normalizedScore":56.0256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-gpqa-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":84.2,"normalizedScore":83.8779,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aagpqadiamond-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.5,"normalizedScore":86.3636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aahle-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":19.7,"normalizedScore":33.0677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aaomniscienceindex-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-46.4,"normalizedScore":32.0251,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-omniscienceaccuracy-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.5,"normalizedScore":29.7251,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-omnisciencehallucinationrate-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84,"normalizedScore":15.6815,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-mmluprox-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":81,"normalizedScore":21.0526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-ifeval-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":91.9,"normalizedScore":90.8392,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-35b-a3b-aaifbench-2026-08-01","modelSlug":"qwen3-5-35b-a3b","modelName":"Qwen3.5-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.5,"normalizedScore":84.6608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-terminalbench2-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":51.5,"normalizedScore":28.1139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-claweval-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":68.7,"normalizedScore":88.2682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-qwenclawbench-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":52.6,"normalizedScore":6.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-qwenwebbench-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1397,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-tau3bench-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":67.2,"normalizedScore":6.2016,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-vitabench-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":35.6,"normalizedScore":62.037,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-deepplanning-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-deepplanning","benchmarkName":"DeepPlanning","benchmarkCategory":"agents","benchmarkOrganisation":"DeepPlanning authors","benchmarkVersion":"2026","score":25.9,"normalizedScore":24.0084,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-toolathlon-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":26.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mcpatlas-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":62.8,"normalizedScore":56.8259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-wideresearch-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-wideresearch","benchmarkName":"WideResearch","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":60.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aaagenticindex-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.41,"normalizedScore":38.4434,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-tau2bench-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.3,"normalizedScore":96.1655,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gdpvalaanormalized-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.6,"normalizedScore":40.5286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gdpvalaa-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1052,"normalizedScore":59.1116,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gertlabs-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":42.65,"normalizedScore":35.9256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-sweverified-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.4,"normalizedScore":68.7845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-swemultilingual-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":67.2,"normalizedScore":37.8109,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-swepro-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":49.5,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-livecodebench-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":80.4,"normalizedScore":79.2593,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-nl2repo-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":29.4,"normalizedScore":8.1481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aacodingindex-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.88,"normalizedScore":48.9753,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aascicode-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.8,"normalizedScore":58.8533,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-lcr-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.7,"normalizedScore":84.148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-critpt-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmmu-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":81.7,"normalizedScore":91.937,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmmupro-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":75.3,"normalizedScore":39.6774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-realworldqa-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":85.3,"normalizedScore":94.38,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-omnidocbench15-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnidocbench15","benchmarkName":"OmniDocBench 1.5","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":89.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-charxiv-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":78,"normalizedScore":62.0098,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-simplevqa-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":58.9,"normalizedScore":10.9375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-ccocr-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ccocr","benchmarkName":"CC-OCR","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":81.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-ai2dtest-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ai2dtest","benchmarkName":"AI2D test split","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":92.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-refcocoavg-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":92,"normalizedScore":95.1923,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-odinw13-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-odinw13","benchmarkName":"ODINW13","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":50.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-videommewithsub-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.6,"normalizedScore":46.1538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-videommenosub-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-videommenosub","benchmarkName":"Video-MME without subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":82.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-videommmu-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"video-mmmu","benchmarkName":"Video-MMMU","benchmarkCategory":"multimodal","benchmarkOrganisation":"Video-MMMU","benchmarkVersion":"2026","score":83.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mlvuavg-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mlvuavg","benchmarkName":"MLVU mean average","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aammmupro-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75,"normalizedScore":83.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmlupro-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.2,"normalizedScore":93.7393,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-supergpqa-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":64.7,"normalizedScore":57.8347,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-ceval-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":90,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-gpqa-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":86,"normalizedScore":86.446,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hle-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":21.4,"normalizedScore":24.0351,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aagpqadiamond-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.1,"normalizedScore":85.7955,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aaomniscienceindex-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-21.4,"normalizedScore":51.6484,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-omniscienceaccuracy-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.9,"normalizedScore":26.9759,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-omnisciencehallucinationrate-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.7,"normalizedScore":57.0567,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aaifbench-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":64.4,"normalizedScore":72.7139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2025-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmt2025","benchmarkName":"Harvard-MIT Mathematics Tournament February 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Harvard and MIT Mathematics Departments","benchmarkVersion":"2025","score":90.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hmmtnov2025-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtnov2025","benchmarkName":"Harvard-MIT Mathematics Tournament November 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2025","score":89.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-hmmtfeb2026-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":83.6,"normalizedScore":81.0765,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-mmanswerbench-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmanswerbench","benchmarkName":"MMAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":78.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-35b-a3b-aime2026-2026-08-01","modelSlug":"qwen3-6-35b-a3b","modelName":"Qwen3.6-35B-A3B","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":92.7,"normalizedScore":88.9418,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.6-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-terminalbench2-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":41,"normalizedScore":9.4306,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-browsecomp-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":52,"normalizedScore":15.8996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-vitabench-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":15.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aaagenticindex-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.39,"normalizedScore":45.681,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-tau2bench-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gertlabs-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.95,"normalizedScore":30.2198,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gdpvalaanormalized-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":48.8987,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gdpvalaa-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1166,"normalizedScore":64.8662,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-sweverified-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.8,"normalizedScore":69.337,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-livecodebench-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":84.9,"normalizedScore":87.5926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-swerebench-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.7,"normalizedScore":72.1519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aacodingindex-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.26,"normalizedScore":53.7527,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aascicode-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.1,"normalizedScore":74.5363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aalivecodebench-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-aalivecodebench","benchmarkName":"Artificial Analysis LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.4,"normalizedScore":41.0256,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-lcr-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64,"normalizedScore":84.5443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-critpt-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.7,"normalizedScore":5.2632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-designarenawebsite-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1251,"normalizedScore":79.0698,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-gpqa-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":85.7,"normalizedScore":86.018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-mmlupro-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":84.3,"normalizedScore":92.4587,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-hle-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":24.8,"normalizedScore":30,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aagpqadiamond-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.9,"normalizedScore":88.3523,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aaomniscienceindex-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-34.6,"normalizedScore":41.2873,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-omniscienceaccuracy-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.3,"normalizedScore":44.8454,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-omnisciencehallucinationrate-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.3,"normalizedScore":8.082,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aaifbench-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":67.9,"normalizedScore":77.8761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-aime2025-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":95.7,"normalizedScore":97.7024,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-frontiermathv2tiers13-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.439,"normalizedScore":2.7404,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-7-frontiermathv2tier4-2026-08-01","modelSlug":"glm-4-7","modelName":"GLM-4.7","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-terminalbench2-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":46.3,"normalizedScore":18.8612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-osworldverified-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":39,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-mcpatlas-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":56.1,"normalizedScore":45.3925,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-toolathlon-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":35.5,"normalizedScore":17.6591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-tau2bench-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":92.5,"normalizedScore":93.3401,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aaagenticindex-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.54,"normalizedScore":49.5908,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-apexagentsaa-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":24.9,"normalizedScore":52.1552,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.1,"normalizedScore":44.1997,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-gdpvalaa-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1101,"normalizedScore":61.5851,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-vibecodebench-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":26.097,"normalizedScore":36.7548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aacodingindex-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.07,"normalizedScore":69.0318,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aascicode-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.9,"normalizedScore":77.5717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-lcr-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66,"normalizedScore":87.1863,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-critpt-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":9.3,"normalizedScore":28.7926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-mmmupro-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":66.1,"normalizedScore":10,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-mmmupropython-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmmupropython","benchmarkName":"MMMU-Pro with Python","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":69.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aammmupro-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":65.4,"normalizedScore":66.8385,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-gpqa-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":82.8,"normalizedScore":81.8804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-hle-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":37.7,"normalizedScore":52.6316,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-hlenotools-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":24.3,"normalizedScore":35.5019,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aagpqadiamond-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81.7,"normalizedScore":82.3864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-29.5,"normalizedScore":45.2904,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.4,"normalizedScore":38.1443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.6,"normalizedScore":28.2268,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-aaifbench-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.9,"normalizedScore":89.6755,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":25.86,"normalizedScore":29.0562,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-4-nano-frontiermathv2tier4-2026-08-01","modelSlug":"gpt-5-4-nano","modelName":"GPT-5.4 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":6.25,"normalizedScore":7.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.4 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-terminalbench2-2026-08-01","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":82.1,"normalizedScore":82.5623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-swepro-2026-08-01","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":73.7,"normalizedScore":82.3529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-livecodebenchv6-2026-08-01","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":93.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-livecodebenchpro-2026-08-01","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":90.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-scicode-2026-08-01","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":58.7,"normalizedScore":95.7704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-mrcrv2-2026-08-01","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":93.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-charxiv-2026-08-01","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":86.6,"normalizedScore":83.0882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-gpqa-2026-08-01","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-gpqadiamond-2026-08-01","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-ultra-hlenotools-2026-08-01","modelSlug":"sakana-fugu-ultra","modelName":"Sakana Fugu-Ultra","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":50,"normalizedScore":83.2714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu-Ultra; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-terminalbench2-2026-08-01","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":80.2,"normalizedScore":79.1815,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-swepro-2026-08-01","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":59,"normalizedScore":43.0481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-livecodebenchv6-2026-08-01","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":92.9,"normalizedScore":99.4973,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-livecodebenchpro-2026-08-01","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":87.8,"normalizedScore":95.596,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-scicode-2026-08-01","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":60.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-mrcrv2-2026-08-01","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":86.6,"normalizedScore":86.0558,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-charxiv-2026-08-01","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":85.1,"normalizedScore":79.4118,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-gpqa-2026-08-01","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-gpqadiamond-2026-08-01","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":95.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-hlenotools-2026-08-01","modelSlug":"sakana-fugu","modelName":"Sakana Fugu","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":47.2,"normalizedScore":78.0669,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sakana Fugu; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-swe-1-7-terminalbench2-2026-08-01","modelSlug":"swe-1-7","modelName":"SWE-1.7","providerId":"cognition","providerName":"Cognition","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":81.5,"normalizedScore":81.4947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-swe-1-7-frontiercode-2026-08-01","modelSlug":"swe-1-7","modelName":"SWE-1.7","providerId":"cognition","providerName":"Cognition","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":42.3,"normalizedScore":61.6438,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-swe-1-7-swemultilingual-2026-08-01","modelSlug":"swe-1-7","modelName":"SWE-1.7","providerId":"cognition","providerName":"Cognition","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":77.8,"normalizedScore":64.1791,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant SWE-1.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-35b-a3b-osworldverified-2026-08-01","modelSlug":"holo3-35b-a3b","modelName":"Holo3-35B-A3B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":82.56,"normalizedScore":94.6957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-122b-a10b-osworldverified-2026-08-01","modelSlug":"holo3-122b-a10b","modelName":"Holo3-122B-A10B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":78.85,"normalizedScore":86.6304,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3-122B-A10B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-tau2bench-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":4.1,"normalizedScore":4.1372,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aascicode-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":25.2,"normalizedScore":40.9781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-lcr-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8,"normalizedScore":10.568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-critpt-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-mmlupro-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":81.8,"normalizedScore":88.9015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aagpqadiamond-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":62.8,"normalizedScore":55.5398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aahle-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.9,"normalizedScore":3.5857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aaomniscienceindex-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-62.3,"normalizedScore":19.5447,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-omniscienceaccuracy-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.4,"normalizedScore":12.3711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-omnisciencehallucinationrate-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81,"normalizedScore":19.3004,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aaifbench-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":33.5,"normalizedScore":27.1386,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-32b-aime2025-2026-08-01","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":85.3,"normalizedScore":79.3213,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-osworldverified-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":83,"normalizedScore":95.6522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-gdpvalaa-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1423,"normalizedScore":77.8395,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aaagenticindex-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.72,"normalizedScore":69.9218,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-gdpvalaanormalized-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.2,"normalizedScore":67.8414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aabriefcaseelo-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":964,"normalizedScore":37.2093,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aatau3banking-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.5,"normalizedScore":53.1579,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aaterminalbench21-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.5,"normalizedScore":65.8824,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aaautomationbench-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.1,"normalizedScore":92,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-deepswe-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":49,"normalizedScore":26.6254,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aacodingindex-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.24,"normalizedScore":87.6466,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aascicode-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":52.7,"normalizedScore":87.3524,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-lcr-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-critpt-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":10.6,"normalizedScore":32.8173,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aammmupro-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":83.2,"normalizedScore":97.4227,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-designarenawebsite-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1319,"normalizedScore":90.3654,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aagpqadiamond-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":92.8,"normalizedScore":98.1534,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aahle-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":38.3,"normalizedScore":70.1195,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-aaomniscienceindex-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.5,"normalizedScore":86.8917,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-omniscienceaccuracy-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.2,"normalizedScore":80.756,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-6-flash-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemini-3-6-flash","modelName":"Gemini 3.6 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":53.5,"normalizedScore":52.4729,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-terminalbench2-2026-08-01","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":77.5,"normalizedScore":74.3772,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-claweval-2026-08-01","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":77.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-sweverified-2026-08-01","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":82.4,"normalizedScore":81.2155,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-swepro-2026-08-01","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":62.2,"normalizedScore":51.6043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-swemultilingual-2026-08-01","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.9,"normalizedScore":66.9154,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-397b-nl2repo-2026-08-01","modelSlug":"ornith-1-0-397b","modelName":"Ornith-1.0-397B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":48.2,"normalizedScore":77.7778,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-397B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-terminalbench2-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":83.3,"normalizedScore":84.6975,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-deepswe-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":53,"normalizedScore":39.0093,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaagenticindex-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.69,"normalizedScore":82.5968,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-gdpvalaanormalized-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.4,"normalizedScore":75.4772,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-gdpvalaa-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1528,"normalizedScore":83.1398,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aabriefcaseelo-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1317,"normalizedScore":66.5282,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaautomationbench-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.4,"normalizedScore":93.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaenterpriseopsgym-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.8,"normalizedScore":59.7656,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaharveylab-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":92.4,"normalizedScore":97.2739,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aatau3banking-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.6,"normalizedScore":95.7895,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaterminalbench21-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.6,"normalizedScore":77.9412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-swepro-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":64.7,"normalizedScore":58.2888,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-swemultilingual-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78,"normalizedScore":64.6766,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-vulcanbench-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vulcanbench","benchmarkName":"VulcanBench v3","benchmarkCategory":"coding","benchmarkOrganisation":"VulcanBench contributors","benchmarkVersion":"2026","score":91.3,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aacodingindex-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.45,"normalizedScore":92.1837,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aascicode-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":54.1,"normalizedScore":89.7133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-arcagi2-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":52.64,"normalizedScore":49.4804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-arcagi3-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.6984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-lcr-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.7,"normalizedScore":89.432,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-critpt-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":15.4,"normalizedScore":47.678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aammmupro-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.4,"normalizedScore":92.6117,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-designarenawebsite-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1321,"normalizedScore":90.6977,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aagpqadiamond-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":93.1,"normalizedScore":98.5795,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aahle-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":40.3,"normalizedScore":74.1036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-aaomniscienceindex-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.4,"normalizedScore":89.168,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-omniscienceaccuracy-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.1,"normalizedScore":84.0206,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-5-omnisciencehallucinationrate-2026-08-01","modelSlug":"grok-4-5","modelName":"Grok 4.5","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":53.5,"normalizedScore":52.4729,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-browsecomp-2026-08-01","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":75.51,"normalizedScore":65.0837,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-hlewithtools-2026-08-01","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":47.6,"normalizedScore":37.3626,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-vitabench-2026-08-01","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":38.75,"normalizedScore":71.7593,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-longbenchv2-2026-08-01","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":60.2,"normalizedScore":78.6802,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-hle-2026-08-01","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":47.6,"normalizedScore":70,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-agents-a1-ifeval-2026-08-01","modelSlug":"agents-a1","modelName":"Agents-A1","providerId":"internscience","providerName":"InternScience","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":94.82,"normalizedScore":99.4681,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Agents-A1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-celeris-1-mmlupro-2026-08-01","modelSlug":"celeris-1","modelName":"Celeris-1","providerId":"celeris","providerName":"Celeris","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":75.9,"normalizedScore":80.5065,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Celeris-1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-tau2bench-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74,"normalizedScore":74.672,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-gertlabs-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":65.59,"normalizedScore":84.4041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-researchclawbench-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":20.7,"normalizedScore":95.4023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-osworld2-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-osworld2","benchmarkName":"OSWorld 2.0","benchmarkCategory":"agents","benchmarkOrganisation":"Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, Haoyuan Wu, Yitong Li, Dunjie Lu, Haikong Lu, Qi Zhen, Xinyuan Wang, Jiaqi Deng, Yuhao Yang, Cheng Chen, Boyuan Zheng, Alex Su, Xiao Yu, Hao Zou, Saaket Agashe, Xing Han Lu, Manpreet Kaur, Zhengyang Qi, Vincent Sunn Chen, Frederic Sala, Dayiheng Liu, Junyang Lin, Zhou Yu, Yu Su, Siva Reddy, Xin Eric Wang, Peng Qi, Tianbao Xie, Tao Yu","benchmarkVersion":"2026","score":13.9,"normalizedScore":16.3717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-vibecodebench-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":71.003,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-reactnativeevals-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":82.8,"normalizedScore":47.012,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aascicode-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50.1,"normalizedScore":82.968,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-frontiercode-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiercode","benchmarkName":"FrontierCode 1.1 Main","benchmarkCategory":"coding","benchmarkOrganisation":"Cognition","benchmarkVersion":"2026","score":38.5,"normalizedScore":48.6301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-lcr-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67,"normalizedScore":88.5073,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-critpt-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.1,"normalizedScore":15.7895,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aammmupro-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":76.4,"normalizedScore":85.7388,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-designarenawebsite-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1320,"normalizedScore":90.5316,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aagpqadiamond-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":88.5,"normalizedScore":92.0455,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aahle-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":31.2,"normalizedScore":55.9761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aaomniscienceindex-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.2,"normalizedScore":79.5918,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-omniscienceaccuracy-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.5,"normalizedScore":69.244,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-omnisciencehallucinationrate-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.9,"normalizedScore":54.4029,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-aaifbench-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43.6,"normalizedScore":42.0354,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-frontiermathv2tiers13-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":43.793,"normalizedScore":49.2056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-7-frontiermathv2tier4-2026-08-01","modelSlug":"claude-opus-4-7","modelName":"Claude Opus 4.7","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":22.917,"normalizedScore":27.6108,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-claweval-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":40.2,"normalizedScore":48.4637,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-vitabench-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":18.5,"normalizedScore":9.2593,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-tau2bench-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":78.9,"normalizedScore":79.6165,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-gertlabs-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":29.57,"normalizedScore":8.284,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-swerebench-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":60.9,"normalizedScore":81.4346,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-reactnativeevals-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71.5,"normalizedScore":1.992,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aascicode-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.7,"normalizedScore":63.7437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-lcr-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39,"normalizedScore":51.5192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-critpt-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-designarenawebsite-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1200,"normalizedScore":70.598,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aagpqadiamond-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":75.1,"normalizedScore":73.0114,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aahle-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":10.5,"normalizedScore":14.741,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aaomniscienceindex-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-46.7,"normalizedScore":31.7896,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-omniscienceaccuracy-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.2,"normalizedScore":36.0825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-omnisciencehallucinationrate-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.5,"normalizedScore":4.222,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-aaifbench-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":49,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-frontiermathv2tiers13-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":22.1,"normalizedScore":24.8315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-frontiermathv2tier4-2026-08-01","modelSlug":"deepseek-v3-2","modelName":"DeepSeek V3.2","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.1,"normalizedScore":2.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-5-terminalbench2-2026-08-01","modelSlug":"composer-2-5","modelName":"Composer 2.5","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":69.3,"normalizedScore":59.7865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-5-swemultilingual-2026-08-01","modelSlug":"composer-2-5","modelName":"Composer 2.5","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":79.8,"normalizedScore":69.1542,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-terminalbench2-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":77.3,"normalizedScore":74.0214,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-osworldverified-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":64.7,"normalizedScore":55.8696,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-tau2bench-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86,"normalizedScore":86.781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-gertlabs-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":57.47,"normalizedScore":67.2443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-jobbench-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":33.7,"normalizedScore":54.5159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-sweverified-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":85,"normalizedScore":84.8066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-swepro-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.8,"normalizedScore":37.1658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-swerebench-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58.2,"normalizedScore":70.0422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-vibecodebench-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":61.767,"normalizedScore":86.9921,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aascicode-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":53.2,"normalizedScore":88.1956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-lcr-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-critpt-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":16.9,"normalizedScore":52.322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aammmupro-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.5,"normalizedScore":89.3471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-designarenawebsite-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1187,"normalizedScore":68.4385,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aagpqadiamond-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":91.5,"normalizedScore":96.3068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aahle-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":39.9,"normalizedScore":73.3068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.9,"normalizedScore":76.2166,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.8,"normalizedScore":83.5052,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.9,"normalizedScore":12.1834,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-3-codex-aaifbench-2026-08-01","modelSlug":"gpt-5-3-codex","modelName":"GPT-5.3-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.4,"normalizedScore":88.9381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.3 Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-tau2bench-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":62.6,"normalizedScore":63.1685,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aacodingindex-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.72,"normalizedScore":45.9223,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aascicode-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.8,"normalizedScore":58.8533,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-lcr-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59.3,"normalizedScore":78.3355,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-critpt-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-mmlu-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":91.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-gpqa-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":75.7,"normalizedScore":71.7506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aagpqadiamond-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":74.7,"normalizedScore":72.4432,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aahle-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7.7,"normalizedScore":9.1633,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aaomniscienceindex-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.5,"normalizedScore":60.2041,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-omniscienceaccuracy-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.7,"normalizedScore":54.1237,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-omnisciencehallucinationrate-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":33.4138,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-ifeval-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":92.2,"normalizedScore":91.7258,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-aaifbench-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.3,"normalizedScore":81.4159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-frontiermathv2tiers13-2026-08-01","modelSlug":"o1","modelName":"o1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":9.31,"normalizedScore":10.4607,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-terminalbench2-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":54,"normalizedScore":32.5623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-osworldverified-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":74,"normalizedScore":76.087,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-gdpvalaa-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1139,"normalizedScore":63.5033,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aaagenticindex-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.82,"normalizedScore":48.2815,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-gdpvalaanormalized-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.9,"normalizedScore":46.8429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aabriefcaseelo-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":636,"normalizedScore":9.9668,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aatau3banking-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.5,"normalizedScore":11.0526,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aaautomationbench-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-automation-bench","benchmarkName":"AA AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-swepro-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":54.2,"normalizedScore":30.2139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aacodingindex-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.32,"normalizedScore":59.4912,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aascicode-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.9,"normalizedScore":67.4536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-mrcrv2-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":72.2,"normalizedScore":57.3705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-lcr-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62,"normalizedScore":81.9022,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-critpt-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aammmupro-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":79,"normalizedScore":90.2062,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aagpqadiamond-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":83.8,"normalizedScore":85.3693,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aahle-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":17.5,"normalizedScore":28.6853,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-aaomniscienceindex-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.9,"normalizedScore":73.8619,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-omniscienceaccuracy-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.3,"normalizedScore":46.5636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-lite-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemini-3-5-flash-lite","modelName":"Gemini 3.5 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.5,"normalizedScore":76.5983,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-bfclv4-2026-08-01","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":39.22,"normalizedScore":33.7039,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-livecodebenchv6-2026-08-01","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":65.8,"normalizedScore":54.0885,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-gpqa-2026-08-01","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":71,"normalizedScore":65.0449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-gpqadiamond-2026-08-01","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":71,"normalizedScore":65.0449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-mmlupro-2026-08-01","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":74.2,"normalizedScore":78.0876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-ifeval-2026-08-01","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":85.58,"normalizedScore":72.1631,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-ifbench-2026-08-01","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":52.56,"normalizedScore":30.3863,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-aime2026-2026-08-01","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":89.1,"normalizedScore":82.8173,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-hmmtfeb2026-2026-08-01","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":71.6,"normalizedScore":64.2557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-imoanswerbench-2026-08-01","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":59.3,"normalizedScore":43.8757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-8b-apex-2026-08-01","modelSlug":"zaya1-8b","modelName":"ZAYA1-8B","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":32.2,"normalizedScore":72.1088,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-terminalbench2-2026-08-01","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":61.7,"normalizedScore":46.2633,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-swemultilingual-2026-08-01","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.7,"normalizedScore":53.9801,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-swerebench-2026-08-01","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":58,"normalizedScore":69.1983,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-reactnativeevals-2026-08-01","modelSlug":"composer-2","modelName":"Composer 2","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":96.1,"normalizedScore":100,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-tau3bench-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":91.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaagenticindex-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19,"normalizedScore":34.0607,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-tau2bench-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.2,"normalizedScore":95.0555,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-gdpvalaanormalized-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.6,"normalizedScore":31.7181,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-gdpvalaa-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":933,"normalizedScore":53.1045,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-gertlabs-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.1,"normalizedScore":28.4235,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaenterpriseopsgym-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.7,"normalizedScore":32.0313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaharveylab-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.1,"normalizedScore":68.4015,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-terminalbenchhard-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":23.1132,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aabriefcaseelo-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":516,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aatau3banking-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-sweverified-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.6,"normalizedScore":74.5856,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aacodingindex-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.9,"normalizedScore":56.0707,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aascicode-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.6,"normalizedScore":65.2614,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-lcr-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61,"normalizedScore":80.5812,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-critpt-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aammmupro-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":64.9,"normalizedScore":65.9794,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aagpqadiamond-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":74.8,"normalizedScore":72.5852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aahle-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":12.8,"normalizedScore":19.3227,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaomniscienceindex-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-36.3,"normalizedScore":39.9529,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-omniscienceaccuracy-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.1,"normalizedScore":37.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-omnisciencehallucinationrate-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82,"normalizedScore":18.0941,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaopennessindex-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.3,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-5-128b-aaifbench-2026-08-01","modelSlug":"mistral-medium-3-5-128b","modelName":"Mistral Medium 3.5 128B","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":68.8,"normalizedScore":79.2035,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3.5 128B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-spider2lite-2026-08-01","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-spider2lite","benchmarkName":"Spider 2.0-Lite","benchmarkCategory":"coding","benchmarkOrganisation":"Spider 2.0 authors","benchmarkVersion":"2024","score":52.9,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-ocrbenchv2-2026-08-01","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"ocrbench-v2","benchmarkName":"OCRBench v2","benchmarkCategory":"multimodal","benchmarkOrganisation":"OCRBench authors","benchmarkVersion":"2025","score":70.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-olmocr-2026-08-01","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-olmocr","benchmarkName":"olmOCR-Bench","benchmarkCategory":"multimodal","benchmarkOrganisation":"Allen Institute for AI","benchmarkVersion":"2025","score":85.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-refcocoavg-2026-08-01","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":82.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-voxpopuliwer-2026-08-01","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-voxpopuliwer","benchmarkName":"VoxPopuli-Cleaned-AA Word Error Rate","benchmarkCategory":"multimodal","benchmarkOrganisation":"Artificial Analysis / VoxPopuli dataset authors","benchmarkVersion":"2026","score":2.4,"normalizedScore":50,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-mmmupro-2026-08-01","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":71.1,"normalizedScore":26.129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-gpqa-2026-08-01","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":89.9,"normalizedScore":92.0103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-gpqadiamond-2026-08-01","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":89.9,"normalizedScore":92.0103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-mmmlu-2026-08-01","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-interfaze-beta-sobvalueacc-2026-08-01","modelSlug":"interfaze-beta","modelName":"Interfaze Beta","providerId":"interfaze","providerName":"Interfaze","benchmarkSlug":"benchlm-sobvalueacc","benchmarkName":"Structured Output Benchmark Value Accuracy","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Interfaze","benchmarkVersion":"2026","score":79.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Interfaze Beta; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-terminalbench2-2026-08-01","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":70.2,"normalizedScore":61.3879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-toolathlonverified-2026-08-01","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-toolathlonverified","benchmarkName":"Toolathlon-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":49.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-swemultilingual-2026-08-01","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":78.5,"normalizedScore":65.9204,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-swepro-2026-08-01","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":59.4,"normalizedScore":44.1176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-s-2-1-deepswe-2026-08-01","modelSlug":"laguna-s-2-1","modelName":"Laguna S 2.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":40.4,"normalizedScore":0,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna S 2.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-terminalbench2-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":65.4,"normalizedScore":52.847,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-qwenclawbench-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenclawbench","benchmarkName":"QwenClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":59,"normalizedScore":57.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-qwenwebbench-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-qwenwebbench","benchmarkName":"QwenWebBench","benchmarkCategory":"agents","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":1532,"normalizedScore":78.9474,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-tau2bench-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95.9,"normalizedScore":96.7709,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-swepro-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.3,"normalizedScore":38.5027,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-scicode-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":47,"normalizedScore":60.423,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-nl2repo-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":42.9,"normalizedScore":58.1481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aascicode-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":46.9,"normalizedScore":77.5717,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-lcr-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-critpt-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":3.7,"normalizedScore":11.4551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-supergpqa-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":73.9,"normalizedScore":70.6374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aagpqadiamond-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":88.8,"normalizedScore":92.4716,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aahle-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.9,"normalizedScore":51.3944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aaomniscienceindex-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.2,"normalizedScore":76.4521,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-omniscienceaccuracy-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.7,"normalizedScore":59.2784,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-omnisciencehallucinationrate-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.2,"normalizedScore":63.6912,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-aaifbench-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":76.6,"normalizedScore":90.708,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-frontiermathv2tiers13-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":23.103,"normalizedScore":25.9584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-6-max-preview-frontiermathv2tier4-2026-08-01","modelSlug":"qwen3-6-max-preview","modelName":"Qwen 3.6 Max (preview)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen 3.6 Max (preview); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-claweval-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":63.8,"normalizedScore":81.4246,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-gdpvalaa-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1265,"normalizedScore":69.8637,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-tau3bench-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-tau3bench","benchmarkName":"τ³-Bench Tool-Agent-User Evaluation","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra Research","benchmarkVersion":"2026","score":72.9,"normalizedScore":28.2946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-terminalbench2-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":68.4,"normalizedScore":58.1851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaagenticindex-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.11,"normalizedScore":52.4459,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-tau2bench-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":94.2,"normalizedScore":95.0555,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-gdpvalaanormalized-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.3,"normalizedScore":56.2408,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-apexagentsaa-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":2.4,"normalizedScore":3.6638,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-gertlabs-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":62.7,"normalizedScore":78.2967,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aabriefcaseelo-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":878,"normalizedScore":30.0664,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaitbench-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.2,"normalizedScore":64.4269,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-terminalbenchhard-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":43.2,"normalizedScore":46.4623,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaterminalbench21-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.2,"normalizedScore":29.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaharveylab-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.3,"normalizedScore":73.6059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-swepro-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":57.2,"normalizedScore":38.2353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aacodingindex-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.19,"normalizedScore":74.8551,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aascicode-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":50.2,"normalizedScore":83.1366,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-lcr-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.3,"normalizedScore":96.8296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-critpt-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4,"normalizedScore":12.3839,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-designarenawebsite-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1309,"normalizedScore":88.7043,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-hle-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":48,"normalizedScore":70.7018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-hlenotools-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":34,"normalizedScore":53.5316,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aagpqadiamond-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86.6,"normalizedScore":89.3466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaomniscienceindex-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.6,"normalizedScore":71.2716,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-omniscienceaccuracy-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.6,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-omnisciencehallucinationrate-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.5,"normalizedScore":87.4548,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaopennessindex-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-pro-aaifbench-2026-08-01","modelSlug":"mimo-v2-5-pro","modelName":"MiMo-V2.5-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":79.9,"normalizedScore":95.5752,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-tau2bench-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":97.7,"normalizedScore":98.5873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gdpvalaanormalized-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.2,"normalizedScore":42.8781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aaagenticindex-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.1,"normalizedScore":43.3352,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-apexagentsaa-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":17,"normalizedScore":35.1293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gdpvalaa-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1084,"normalizedScore":60.7269,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gertlabs-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":43.86,"normalizedScore":38.4827,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-researchclawbench-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":12.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-scicode-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":47.3,"normalizedScore":61.3293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aacodingindex-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.25,"normalizedScore":49.4982,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aascicode-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.3,"normalizedScore":78.2462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-lcr-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.3,"normalizedScore":84.9406,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-critpt-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":8,"normalizedScore":24.7678,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-mmmupro-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":78.1,"normalizedScore":48.7097,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-designarenawebsite-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1219,"normalizedScore":73.7542,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aammmupro-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.1,"normalizedScore":88.6598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-gpqa-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":90.1,"normalizedScore":92.2956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-hle-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":35,"normalizedScore":47.8947,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-omniscienceaccuracy-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.6,"normalizedScore":53.9519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-omnisciencehallucinationrate-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25,"normalizedScore":86.8516,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aagpqadiamond-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":90.1,"normalizedScore":94.3182,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aaomniscienceindex-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.3,"normalizedScore":82.81,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-ifbench-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":81.3,"normalizedScore":92.0601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-3-aaifbench-2026-08-01","modelSlug":"grok-4-3","modelName":"Grok 4.3","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":81.3,"normalizedScore":97.6401,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-bigcodebench-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"bigcodebench","benchmarkName":"BigCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"BigCodeBench authors","benchmarkVersion":"2026","score":59.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-humaneval-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-humaneval","benchmarkName":"Evaluating Large Language Models Trained on Code","benchmarkCategory":"coding","benchmarkOrganisation":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","benchmarkVersion":"2021","score":76.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-bbh-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":87.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-drop-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-drop","benchmarkName":"Discrete Reasoning Over Paragraphs","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-hellaswag-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hellaswag","benchmarkName":"HellaSwag","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-winogrande-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-winogrande","benchmarkName":"WinoGrande","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":81.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-cluewsc-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cluewsc","benchmarkName":"CLUEWSC","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-longbenchv2-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":51.5,"normalizedScore":34.5178,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-agieval-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-agieval","benchmarkName":"AGIEval","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":83.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmlu-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":90.1,"normalizedScore":85.4701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmluredux-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":90.8,"normalizedScore":78.1462,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmlupro-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":73.5,"normalizedScore":77.0916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mmmlu-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":90.3,"normalizedScore":92,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-ceval-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":93.1,"normalizedScore":93.9394,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-cmmlu-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmmlu","benchmarkName":"Chinese Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-multiloko-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-multiloko","benchmarkName":"MultiLoKo","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":51.1,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-simpleqa-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":55.2,"normalizedScore":92.2414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-supergpqa-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":53.9,"normalizedScore":42.8055,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-factsparametric-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-factsparametric","benchmarkName":"FACTS Parametric","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":62.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-triviaqa-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-triviaqa","benchmarkName":"TriviaQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mgsm-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mgsm","benchmarkName":"Multilingual Grade School Math","benchmarkCategory":"knowledge","benchmarkOrganisation":"Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","benchmarkVersion":"2022","score":84.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-gsm8k-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gsm8k","benchmarkName":"Grade School Math 8K","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":92.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-mathbenchmark-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mathbenchmark","benchmarkName":"MATH","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":64.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-pro-base-cmath-2026-08-01","modelSlug":"deepseek-v4-pro-base","modelName":"DeepSeek V4 Pro Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmath","benchmarkName":"CMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Pro Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-jobbench-2026-08-01","modelSlug":"claude-4-1-opus","modelName":"Claude 4.1 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":21.9,"normalizedScore":28.9582,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-sweverified-2026-08-01","modelSlug":"claude-4-1-opus","modelName":"Claude 4.1 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.5,"normalizedScore":70.3039,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-designarenawebsite-2026-08-01","modelSlug":"claude-4-1-opus","modelName":"Claude 4.1 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1202,"normalizedScore":70.9302,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-pro-deep-think-arcagi2-2026-08-01","modelSlug":"gemini-3-pro-deep-think","modelName":"Gemini 3 Pro Deep Think","providerId":"google","providerName":"Google","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":45.1,"normalizedScore":39.924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-gemini-3-pro-deep-think-critpt-2026-08-01","modelSlug":"gemini-3-pro-deep-think","modelName":"Gemini 3 Pro Deep Think","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":25.7,"normalizedScore":79.5666,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Pro Deep Think; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-muse-spark-terminalbench2-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":59,"normalizedScore":41.4591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-tau2bench-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":91.5,"normalizedScore":92.331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-deepsearchqa-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":74.8,"normalizedScore":37.2671,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-cybergym-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":43.5,"normalizedScore":0.6865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-claweval-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":63.8,"normalizedScore":81.4246,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aaagenticindex-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.69,"normalizedScore":51.6821,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-gdpvalaanormalized-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.2,"normalizedScore":47.2834,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-gdpvalaa-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1143,"normalizedScore":63.7052,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-sweverified-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.4,"normalizedScore":74.3094,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-swepro-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.4,"normalizedScore":25.4011,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-livecodebenchpro-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":80,"normalizedScore":84.1456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-vibecodebench-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":19.674,"normalizedScore":27.7087,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aacodingindex-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.62,"normalizedScore":72.636,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aascicode-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":51.5,"normalizedScore":85.3288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-arcagi2-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":42.5,"normalizedScore":36.6286,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-lcr-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.7,"normalizedScore":92.074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-critpt-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":11.3,"normalizedScore":34.9845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-charxiv-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":86.4,"normalizedScore":82.598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-mmmupro-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":80.4,"normalizedScore":56.129,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-erqa-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":64.7,"normalizedScore":71.978,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-simplevqa-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":71.3,"normalizedScore":59.375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-screenspotpro-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":84.1,"normalizedScore":90.9953,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-zerobench-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"2026","score":33,"normalizedScore":55.5556,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-medxpertqamm-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":78.4,"normalizedScore":91.1043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aammmupro-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":80.5,"normalizedScore":92.7835,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-gpqadiamond-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":89.5,"normalizedScore":91.4396,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-hle-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":50.4,"normalizedScore":74.9123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-hlenotools-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":42.8,"normalizedScore":69.8885,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-healthbenchhard-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":42.8,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-medxpertqatext-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":52.6,"normalizedScore":11.2676,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aaomniscienceindex-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.1,"normalizedScore":71.6641,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-omniscienceaccuracy-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.6,"normalizedScore":71.134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-omnisciencehallucinationrate-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73.2,"normalizedScore":28.7093,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-aaifbench-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.9,"normalizedScore":89.6755,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-frontiermathv2tiers13-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":39,"normalizedScore":43.8202,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-muse-spark-frontiermathv2tier4-2026-08-01","modelSlug":"muse-spark","modelName":"Muse Spark","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.6,"normalizedScore":17.5904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Muse Spark; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-claweval-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":49.2,"normalizedScore":61.0335,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-tau2bench-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":43.3,"normalizedScore":43.6932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-gertlabs-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":56.63,"normalizedScore":65.4691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-jobbench-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":11.4,"normalizedScore":6.2162,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-vibecodebench-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":20.204,"normalizedScore":28.4551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aascicode-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49.9,"normalizedScore":82.6307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-lcr-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48,"normalizedScore":63.4082,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-critpt-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aammmupro-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":78.6,"normalizedScore":89.5189,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-designarenawebsite-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1221,"normalizedScore":74.0864,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aagpqadiamond-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81.2,"normalizedScore":81.6761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aahle-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.1,"normalizedScore":21.9124,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aaomniscienceindex-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-3.6,"normalizedScore":65.6201,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-omniscienceaccuracy-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.5,"normalizedScore":72.6804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":90.2,"normalizedScore":8.2027,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-aaifbench-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":55.1,"normalizedScore":58.9971,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-frontiermathv2tiers13-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":35.64,"normalizedScore":40.0449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-flash-frontiermathv2tier4-2026-08-01","modelSlug":"gemini-3-flash","modelName":"Gemini 3 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aaagenticindex-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.01,"normalizedScore":37.7159,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-tau2bench-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":81.9,"normalizedScore":82.6438,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-gertlabs-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":41.24,"normalizedScore":32.9459,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.4,"normalizedScore":35.8297,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-gdpvalaa-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":988,"normalizedScore":55.8809,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-vibecodebench-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":24.606,"normalizedScore":34.6549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aacodingindex-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.39,"normalizedScore":59.5901,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aascicode-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.3,"normalizedScore":71.5008,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-lcr-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75,"normalizedScore":99.0753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-critpt-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.9,"normalizedScore":15.1703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aammmupro-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.5,"normalizedScore":84.1924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-designarenawebsite-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1212,"normalizedScore":72.5914,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aagpqadiamond-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87.3,"normalizedScore":90.3409,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aahle-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":26.5,"normalizedScore":46.6135,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.6,"normalizedScore":72.8414,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.6,"normalizedScore":59.1065,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51.3,"normalizedScore":55.1267,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-aaifbench-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.9,"normalizedScore":85.2507,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":31.034,"normalizedScore":34.8697,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-frontiermathv2tier4-2026-08-01","modelSlug":"gpt-5-1","modelName":"GPT-5.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":12.5,"normalizedScore":15.0602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-terminalbench2-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":56.9,"normalizedScore":37.7224,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-browsecomp-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":73.2,"normalizedScore":60.251,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-hlewithtools-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":45.1,"normalizedScore":28.2051,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-mcpatlas-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":69,"normalizedScore":67.4061,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gdpvalaa-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1189,"normalizedScore":66.0273,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-toolathlon-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":47.8,"normalizedScore":42.9158,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-terminalbench21-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-21-provider","benchmarkName":"Terminal-Bench 2.1 (provider run)","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":82.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-cybergym-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":76.7,"normalizedScore":76.659,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-toolathlonverified-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-toolathlonverified","benchmarkName":"Toolathlon-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":70.3,"normalizedScore":66.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-agentslastexam-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-alebench","benchmarkName":"Agents Last Exam","benchmarkCategory":"knowledge","benchmarkOrganisation":"UC Berkeley RDI","benchmarkVersion":"2026","score":25.2,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-automationbench-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-automationbench","benchmarkName":"AutomationBench","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":25.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaagenticindex-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.06,"normalizedScore":55.992,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-tau2bench-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95,"normalizedScore":95.8628,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gdpvalaanormalized-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.4,"normalizedScore":50.514,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aabriefcaseelo-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aabriefcaseelo","benchmarkName":"Artificial Analysis Briefcase","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":833,"normalizedScore":26.3289,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaenterpriseopsgym-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.6,"normalizedScore":55.0781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaharveylab-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.3,"normalizedScore":83.5192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaitbench-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.5,"normalizedScore":51.1858,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aatau3banking-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.9,"normalizedScore":44.7368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-terminalbenchhard-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":35.6,"normalizedScore":28.5377,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaterminalbench21-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaterminalbench21","benchmarkName":"Artificial Analysis Terminal-Bench v2.1","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61.8,"normalizedScore":19.7059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-livecodebenchpass1cot-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-livecodebenchpass1cot","benchmarkName":"LiveCodeBench Pass@1 with Chain-of-Thought","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek","benchmarkVersion":"2026","score":91.6,"normalizedScore":95.0392,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-codeforces-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-codeforces","benchmarkName":"Codeforces Rating","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":3052,"normalizedScore":60.5128,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-sweverified-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":79,"normalizedScore":76.5193,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-swepro-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":52.6,"normalizedScore":25.9358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-swemultilingual-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":73.3,"normalizedScore":52.9851,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-nl2repo-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":54.2,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-deepswe-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"2026","score":54.4,"normalizedScore":43.3437,"unit":"USD","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-dsbenchfullstack-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-dsbenchfullstack","benchmarkName":"DeepSeek DSBench FullStack","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":68.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-dsbenchhard-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-dsbenchhard","benchmarkName":"DeepSeek DSBench Hard","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":59.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aacodingindex-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":56.17,"normalizedScore":69.1731,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aascicode-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":44.9,"normalizedScore":74.199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-mrcr1m-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":78.7,"normalizedScore":91.5641,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-corpusqa1m-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":60.5,"normalizedScore":96.7742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-lcr-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63,"normalizedScore":83.2232,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-critpt-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":7.1,"normalizedScore":21.9814,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-designarenawebsite-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1233,"normalizedScore":76.0797,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-mmlupro-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":86.2,"normalizedScore":95.1622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-simpleqa-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":34.1,"normalizedScore":31.6092,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-chinesesimpleqa-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":78.9,"normalizedScore":57.3643,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gpqa-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":88.1,"normalizedScore":89.4421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-gpqadiamond-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":88.1,"normalizedScore":89.4421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-hle-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":34.8,"normalizedScore":47.5439,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaomniscienceindex-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-22.9,"normalizedScore":50.471,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-omniscienceaccuracy-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.2,"normalizedScore":58.4192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-omnisciencehallucinationrate-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":95.8,"normalizedScore":1.4475,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaopennessindex-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50,"normalizedScore":33.4,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-aaifbench-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":79.2,"normalizedScore":94.5428,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-hmmtfeb2026-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":94.8,"normalizedScore":96.776,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-imoanswerbench-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88.4,"normalizedScore":97.075,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-apex-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":33,"normalizedScore":73.9229,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-max-apexshortlist-2026-08-01","modelSlug":"deepseek-v4-flash-max","modelName":"DeepSeek V4 Flash (Max)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.7,"normalizedScore":94.4444,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash (Max); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-flash-aaagenticindex-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":11.99,"normalizedScore":21.313,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-tau2bench-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83.9,"normalizedScore":84.662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-gdpvalaanormalized-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.9,"normalizedScore":24.8164,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-gdpvalaa-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":838,"normalizedScore":48.3089,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-sweverified-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.4,"normalizedScore":68.7845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aacodingindex-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":49.84,"normalizedScore":60.2261,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aascicode-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":25.9,"normalizedScore":42.1585,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-lcr-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.3,"normalizedScore":41.3474,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-critpt-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-gpqa-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":83.7,"normalizedScore":83.1645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-mmlupro-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":84.9,"normalizedScore":93.3125,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aagpqadiamond-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":65.6,"normalizedScore":59.517,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aahle-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":8,"normalizedScore":9.761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aaomniscienceindex-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-48.5,"normalizedScore":30.3768,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-omniscienceaccuracy-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.2,"normalizedScore":20.6186,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-omnisciencehallucinationrate-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.1,"normalizedScore":26.4174,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aaifbench-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.9,"normalizedScore":36.5782,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-flash-aime2025-2026-08-01","modelSlug":"mimo-v2-flash","modelName":"MiMo-V2-Flash","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":94.1,"normalizedScore":94.8745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"excluded","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"excluded"},{"resultId":"benchlm-ref-mimo-v2-5-claweval-2026-08-01","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":62.3,"normalizedScore":79.3296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-mmclawbench-2026-08-01","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-mmclawbench","benchmarkName":"MM-ClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":23.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-terminalbench2-2026-08-01","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":65.8,"normalizedScore":53.5587,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-gertlabs-2026-08-01","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":46.89,"normalizedScore":44.8859,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-researchclawbench-2026-08-01","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":16.9,"normalizedScore":51.7241,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-swepro-2026-08-01","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.1,"normalizedScore":35.2941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-videommewithsub-2026-08-01","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-videommewithsub","benchmarkName":"Video-MME with subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":87.7,"normalizedScore":88.4615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-charxiv-2026-08-01","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":81,"normalizedScore":69.3627,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-mmmupro-2026-08-01","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":77.9,"normalizedScore":48.0645,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-5-designarenawebsite-2026-08-01","modelSlug":"mimo-v2-5","modelName":"MiMo-V2.5","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1295,"normalizedScore":86.3787,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-tau2bench-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":52.3,"normalizedScore":52.775,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-gertlabs-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":39.66,"normalizedScore":29.6069,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-jobbench-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":18.4,"normalizedScore":21.3775,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-sweverified-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":72.7,"normalizedScore":67.8177,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aascicode-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.3,"normalizedScore":61.3828,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-lcr-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.3,"normalizedScore":58.5205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-critpt-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aammmupro-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":62.4,"normalizedScore":61.6838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-designarenawebsite-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1170,"normalizedScore":65.6146,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aagpqadiamond-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68.3,"normalizedScore":63.3523,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aahle-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4,"normalizedScore":1.7928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aaomniscienceindex-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-9.2,"normalizedScore":61.2245,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-omniscienceaccuracy-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.4,"normalizedScore":32.9897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-omnisciencehallucinationrate-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.8,"normalizedScore":67.7925,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-sonnet-aaifbench-2026-08-01","modelSlug":"claude-4-sonnet","modelName":"Claude 4 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":45.4,"normalizedScore":44.6903,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-terminalbench2-2026-08-01","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":64.2,"normalizedScore":50.7117,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-claweval-2026-08-01","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":69.8,"normalizedScore":89.8045,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-sweverified-2026-08-01","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":75.6,"normalizedScore":71.8232,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-swepro-2026-08-01","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":50.4,"normalizedScore":20.0535,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-swemultilingual-2026-08-01","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":69.3,"normalizedScore":43.0348,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-35b-nl2repo-2026-08-01","modelSlug":"ornith-1-0-35b","modelName":"Ornith-1.0-35B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":34.6,"normalizedScore":27.4074,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-35B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aaagenticindex-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.17,"normalizedScore":10.7292,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-apexagentsaa-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":12.2,"normalizedScore":24.7845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-tau2bench-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":31.3,"normalizedScore":31.5843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-gdpvalaanormalized-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.3,"normalizedScore":10.7195,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-gdpvalaa-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":647,"normalizedScore":38.6673,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-gertlabs-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":38.46,"normalizedScore":27.071,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-vibecodebench-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aacodingindex-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.69,"normalizedScore":38.8127,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aascicode-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":41.9,"normalizedScore":69.14,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-lcr-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":65.3,"normalizedScore":86.2616,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-critpt-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-charxiv-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":73.2,"normalizedScore":50.2451,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aammmupro-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.5,"normalizedScore":84.1924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aagpqadiamond-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":82.2,"normalizedScore":83.0966,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aahle-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":16.2,"normalizedScore":26.0956,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aaomniscienceindex-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-15.5,"normalizedScore":56.2794,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-omniscienceaccuracy-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":36.4,"normalizedScore":57.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.6,"normalizedScore":18.5766,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-1-flash-lite-aaifbench-2026-08-01","modelSlug":"gemini-3-1-flash-lite","modelName":"Gemini 3.1 Flash-Lite","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":77.2,"normalizedScore":91.5929,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.1 Flash-Lite; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-osworld-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"osworld","benchmarkName":"OSWorld","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld authors","benchmarkVersion":"2026","score":47.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-tau2bench-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":45.3,"normalizedScore":45.7114,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaanormalized-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gdpvalaa-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":465,"normalizedScore":29.4801,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-livecodebenchv5-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-livecodebenchv5","benchmarkName":"LiveCodeBench v5","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2025","score":63.2,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-scicode-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":32,"normalizedScore":15.1057,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aascicode-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":27.8,"normalizedScore":45.3626,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aacodingindex-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.75,"normalizedScore":9.2155,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-lcr-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.7,"normalizedScore":47.1598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-critpt-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmmu-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":70.8,"normalizedScore":71.4982,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlongbenchdoc-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmlongbenchdoc","benchmarkName":"MMLongBench-Doc","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":57.5,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-charxiv-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":76.25,"normalizedScore":57.7206,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-screenspotpro-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":57.8,"normalizedScore":28.673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-videommenosub-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-videommenosub","benchmarkName":"Video-MME without subtitle","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":72.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-ai2dtest-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-ai2dtest","benchmarkName":"AI2D test split","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":88.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-refcocoavg-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-refcocoavg","benchmarkName":"RefCOCO average","benchmarkCategory":"multimodal","benchmarkOrganisation":"RefCOCO dataset authors","benchmarkVersion":"2026","score":90.5,"normalizedScore":80.7692,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aammmupro-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":53.2,"normalizedScore":45.8763,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-mmlupro-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":77.3,"normalizedScore":82.4986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqa-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":72.2,"normalizedScore":66.757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-gpqadiamond-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":72.2,"normalizedScore":66.757,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aahle-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.3,"normalizedScore":4.3825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaomniscienceindex-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-56,"normalizedScore":24.4898,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-omniscienceaccuracy-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.8,"normalizedScore":19.9313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-omnisciencehallucinationrate-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.1,"normalizedScore":16.7672,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-ifbench-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":74.2,"normalizedScore":76.824,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aaifbench-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":63.2,"normalizedScore":70.944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-omni-30b-a3b-aime2025-2026-08-01","modelSlug":"nemotron-3-nano-omni-30b-a3b","modelName":"Nemotron 3 Nano Omni 30B A3B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":82.1,"normalizedScore":73.6656,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano Omni 30B A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-jobbench-2026-08-01","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":16,"normalizedScore":16.1793,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-sweverified-2026-08-01","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.3,"normalizedScore":68.6464,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-vulcanbench-2026-08-01","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vulcanbench","benchmarkName":"VulcanBench v3","benchmarkCategory":"coding","benchmarkOrganisation":"VulcanBench contributors","benchmarkVersion":"2026","score":78.26,"normalizedScore":25.0144,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-designarenawebsite-2026-08-01","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1147,"normalizedScore":61.794,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-frontiermathv2tiers13-2026-08-01","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":5.903,"normalizedScore":6.6326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-frontiermathv2tier4-2026-08-01","modelSlug":"claude-haiku-4-5","modelName":"Claude Haiku 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o4-mini-high-frontiermathv2tiers13-2026-08-01","modelSlug":"o4-mini-high","modelName":"o4-mini (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":24.828,"normalizedScore":27.8966,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o4-mini-high-frontiermathv2tier4-2026-08-01","modelSlug":"o4-mini-high","modelName":"o4-mini (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":6.25,"normalizedScore":7.5301,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o4-mini (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-gpqa-2026-08-01","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":77.5,"normalizedScore":74.3187,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-supergpqa-2026-08-01","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":62.6,"normalizedScore":54.9123,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-mmlupro-2026-08-01","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":83,"normalizedScore":90.609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-mmluprox-2026-08-01","modelSlug":"qwen3-235b-2507","modelName":"Qwen3 235B 2507","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-mmluprox","benchmarkName":"MMLU-ProX","benchmarkCategory":"knowledge","benchmarkOrganisation":"MMLU-ProX authors","benchmarkVersion":"2025","score":79.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-terminalbench2-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":59.5,"normalizedScore":42.3488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-browsecomp-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"browsecomp","benchmarkName":"BrowseComp","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":75.82,"normalizedScore":65.7322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-deepsearchqa-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":92.82,"normalizedScore":93.2298,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-gdpvalaanormalized-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.8,"normalizedScore":37.8855,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-toolathlon-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":49.5,"normalizedScore":46.4066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-claweval-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":67.1,"normalizedScore":86.0335,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-hlewithtools-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-hlewithtools","benchmarkName":"Humanity's Last Exam with tools","benchmarkCategory":"agents","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":47.2,"normalizedScore":35.8974,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-gertlabs-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":51.57,"normalizedScore":54.776,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aaagenticindex-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.53,"normalizedScore":38.6616,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-tau2bench-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-gdpvalaa-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1017,"normalizedScore":57.3448,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-apexagentsaa-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":14.8,"normalizedScore":30.3879,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-swepro-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.3,"normalizedScore":35.8289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aacodingindex-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.57,"normalizedScore":45.7102,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aascicode-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40,"normalizedScore":65.9359,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-lcr-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":63.7,"normalizedScore":84.148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-critpt-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.3,"normalizedScore":7.1207,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-simplevqa-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":79.2,"normalizedScore":90.2344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-vstar-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-vstar","benchmarkName":"V*","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":95.3,"normalizedScore":94.6488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aammmupro-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":75.3,"normalizedScore":83.8488,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-designarenawebsite-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1212,"normalizedScore":72.5914,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aagpqadiamond-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":80.9,"normalizedScore":81.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aahle-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":19.9,"normalizedScore":33.4661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aaomniscienceindex-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-37.5,"normalizedScore":39.011,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-omniscienceaccuracy-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.4,"normalizedScore":38.1443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-omnisciencehallucinationrate-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84.4,"normalizedScore":15.199,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-step-3-7-flash-aaifbench-2026-08-01","modelSlug":"step-3-7-flash","modelName":"Step 3.7 Flash","providerId":"stepfun","providerName":"StepFun","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":67.3,"normalizedScore":76.9912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Step 3.7 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-bfclv4-2026-08-01","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":45.6,"normalizedScore":45.5253,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-livecodebenchv6-2026-08-01","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":69.9,"normalizedScore":60.9584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-mmluredux-2026-08-01","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":86.2,"normalizedScore":60.8139,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-gpqa-2026-08-01","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":57.6,"normalizedScore":45.9267,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-gpqadiamond-2026-08-01","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":57.6,"normalizedScore":45.9267,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-thinking-ifeval-2026-08-01","modelSlug":"mellum2-12b-a2-5b-thinking","modelName":"Mellum2-12B-A2.5B-Thinking","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":76.5,"normalizedScore":45.331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-tau2bench-2026-08-01","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":28.7,"normalizedScore":28.9606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-sweverified-2026-08-01","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":49.3,"normalizedScore":35.4972,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aascicode-2026-08-01","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":39.9,"normalizedScore":65.7673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-mmlu-2026-08-01","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":86.9,"normalizedScore":58.1197,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-gpqa-2026-08-01","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":77.2,"normalizedScore":73.8907,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aagpqadiamond-2026-08-01","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":74.8,"normalizedScore":72.5852,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aahle-2026-08-01","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":8.7,"normalizedScore":11.1554,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-ifeval-2026-08-01","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":93.9,"normalizedScore":96.7494,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-mini-aime2024-2026-08-01","modelSlug":"o3-mini","modelName":"o3-mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aime2024","benchmarkName":"American Invitational Mathematics Examination 2024","benchmarkCategory":"mathematics","benchmarkOrganisation":"Mathematical Association of America","benchmarkVersion":"2024","score":87.3,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-jobbench-2026-08-01","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":18.5,"normalizedScore":21.5941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-vibecodebench-2026-08-01","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":15.738,"normalizedScore":22.1653,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-frontiermathv2tiers13-2026-08-01","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":21.034,"normalizedScore":23.6337,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-plus-frontiermathv2tier4-2026-08-01","modelSlug":"qwen3-5-plus","modelName":"Qwen3.5 Plus","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Plus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-bfclv4-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":49.73,"normalizedScore":53.1777,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-tau2bench-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":88.07,"normalizedScore":88.8698,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aascicode-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":7.8,"normalizedScore":11.6358,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-lcr-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-critpt-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aagpqadiamond-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":46.6,"normalizedScore":32.5284,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aahle-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.9,"normalizedScore":7.5697,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aaomniscienceindex-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-33.3,"normalizedScore":42.3077,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-omniscienceaccuracy-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.4,"normalizedScore":10.6529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-omnisciencehallucinationrate-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47,"normalizedScore":60.3136,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-ifeval-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":91.84,"normalizedScore":90.6619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-ifbench-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":56.47,"normalizedScore":38.7768,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aaifbench-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":53.3,"normalizedScore":56.3422,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-math500-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-math500","benchmarkName":"MATH-500 Problem Set","benchmarkCategory":"mathematics","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2021","score":88.76,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aime2025-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":42.53,"normalizedScore":3.7292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-8b-a1b-aime2026-2026-08-01","modelSlug":"lfm2-5-8b-a1b","modelName":"LFM2.5-8B-A1B","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":50,"normalizedScore":16.2981,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-8B-A1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-terminalbench2-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":57,"normalizedScore":37.9004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-tau2bench-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-toolathlon-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":46.3,"normalizedScore":39.8357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-mlebenchlite-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mlebenchlite","benchmarkName":"MLE-Bench Lite","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":66.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-mmclawbench-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mmclawbench","benchmarkName":"MM-ClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":62.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-claweval-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":48.7,"normalizedScore":60.3352,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aaagenticindex-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.58,"normalizedScore":46.0266,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-apexagentsaa-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":10.6,"normalizedScore":21.3362,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gdpvalaanormalized-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.9,"normalizedScore":48.3113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gdpvalaa-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1159,"normalizedScore":64.5129,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gertlabs-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":40.4,"normalizedScore":31.1708,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-sweverifiedarcee-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75.4,"normalizedScore":98.3871,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-swepro-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":56.2,"normalizedScore":35.5615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-swerebench-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":51.9,"normalizedScore":43.4599,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-swemultilingual-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":76.5,"normalizedScore":60.9453,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-multiswebench-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"multi-swe-bench","benchmarkName":"Multi-SWE-Bench","benchmarkCategory":"coding","benchmarkOrganisation":"Multi-SWE-Bench","benchmarkVersion":"2026","score":52.7,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-vibepro-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibepro","benchmarkName":"VIBE-Pro","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":55.6,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-nl2repo-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":39.8,"normalizedScore":46.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-vibecodebench-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":27.037,"normalizedScore":38.0787,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-reactnativeevals-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71.4,"normalizedScore":1.5936,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aacodingindex-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":52.62,"normalizedScore":64.1555,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aascicode-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47,"normalizedScore":77.7403,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-lcr-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68.7,"normalizedScore":90.753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-critpt-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-designarenawebsite-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1271,"normalizedScore":82.392,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-gpqadiamond-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87,"normalizedScore":87.8727,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-mmluproarcee-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":80.8,"normalizedScore":40.2878,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aahle-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.1,"normalizedScore":49.8008,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aaomniscienceindex-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.7,"normalizedScore":68.9953,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-omniscienceaccuracy-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.1,"normalizedScore":39.3471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-omnisciencehallucinationrate-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.4,"normalizedScore":75.5127,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aaifbench-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.7,"normalizedScore":89.3805,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-7-aime2025arcee-2026-08-01","modelSlug":"minimax-m2-7","modelName":"MiniMax M2.7","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":80,"normalizedScore":73.8786,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.7; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-tau2bench-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":61.1,"normalizedScore":61.6549,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aascicode-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":34.5,"normalizedScore":56.661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-lcr-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51,"normalizedScore":67.3712,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-critpt-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-designarenawebsite-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1075,"normalizedScore":49.8339,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aagpqadiamond-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.6,"normalizedScore":75.142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aahle-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7,"normalizedScore":7.7689,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aaomniscienceindex-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-27.5,"normalizedScore":46.8603,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-omniscienceaccuracy-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":40.5498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-omnisciencehallucinationrate-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.2,"normalizedScore":27.503,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-aaifbench-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":41.5,"normalizedScore":38.9381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-frontiermathv2tiers13-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":21.404,"normalizedScore":24.0494,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-frontiermathv2tier4-2026-08-01","modelSlug":"kimi-k2","modelName":"Kimi K2","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-tau2bench-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74.9,"normalizedScore":75.5802,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-gertlabs-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":42.34,"normalizedScore":35.2705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-reactnativeevals-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":72.6,"normalizedScore":6.3745,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aascicode-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":45.7,"normalizedScore":75.5481,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-lcr-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68,"normalizedScore":89.8283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-critpt-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2,"normalizedScore":6.192,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aammmupro-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":68.8,"normalizedScore":72.6804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aagpqadiamond-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87.7,"normalizedScore":90.9091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aahle-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.9,"normalizedScore":41.4343,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aaomniscienceindex-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.8,"normalizedScore":71.4286,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-omniscienceaccuracy-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":41.4,"normalizedScore":65.6357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-omnisciencehallucinationrate-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.2,"normalizedScore":39.5657,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-aaifbench-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":53.7,"normalizedScore":56.9322,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-frontiermathv2tiers13-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":19.655,"normalizedScore":22.0843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-frontiermathv2tier4-2026-08-01","modelSlug":"grok-4","modelName":"Grok 4","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-humaneval-2026-08-01","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-humaneval","benchmarkName":"Evaluating Large Language Models Trained on Code","benchmarkCategory":"coding","benchmarkOrganisation":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","benchmarkVersion":"2021","score":73.8,"normalizedScore":58.9041,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-bbh-2026-08-01","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":78.8,"normalizedScore":74.7826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-drop-2026-08-01","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-drop","benchmarkName":"Discrete Reasoning Over Paragraphs","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":66.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-gpqa-2026-08-01","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":43.4,"normalizedScore":25.667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-gpqadiamond-2026-08-01","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":43.4,"normalizedScore":25.667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-mmlupro-2026-08-01","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":51.4,"normalizedScore":45.646,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-agieval-2026-08-01","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-agieval","benchmarkName":"AGIEval","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":66.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-soofi-s-30b-a3b-gsm8k-2026-08-01","modelSlug":"soofi-s-30b-a3b","modelName":"Soofi S 30B-A3B","providerId":"soofi-project","providerName":"Soofi Project","benchmarkSlug":"benchlm-gsm8k","benchmarkName":"Grade School Math 8K","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":86.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Soofi S 30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-tau2bench-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":80.7,"normalizedScore":81.4329,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aascicode-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":41,"normalizedScore":67.6223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-lcr-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":69.3,"normalizedScore":91.5456,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-critpt-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aammmupro-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":70.1,"normalizedScore":74.9141,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-designarenawebsite-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1061,"normalizedScore":47.5083,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aagpqadiamond-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":82.7,"normalizedScore":83.8068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aahle-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":20,"normalizedScore":33.6653,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aaomniscienceindex-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-15.3,"normalizedScore":56.4364,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-omniscienceaccuracy-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.4,"normalizedScore":60.4811,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-omnisciencehallucinationrate-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.1,"normalizedScore":11.9421,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aaifbench-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":71.4,"normalizedScore":83.0383,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-aamath500-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aamath500","benchmarkName":"Artificial Analysis MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99.2,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-frontiermathv2tiers13-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":18.685,"normalizedScore":20.9944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-frontiermathv2tier4-2026-08-01","modelSlug":"o3","modelName":"o3","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.083,"normalizedScore":2.5096,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aaagenticindex-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.97,"normalizedScore":19.4581,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-tau2bench-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":43.6,"normalizedScore":43.996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-gdpvalaanormalized-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.5,"normalizedScore":19.8238,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-gdpvalaa-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":770,"normalizedScore":44.8763,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aacodingindex-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.32,"normalizedScore":45.3569,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aascicode-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40,"normalizedScore":65.9359,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-lcr-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.7,"normalizedScore":73.5799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-critpt-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-mmmupro-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":73.8,"normalizedScore":34.8387,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aammmupro-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":69.2,"normalizedScore":73.3677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-mmlupro-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":82.6,"normalizedScore":90.0398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-hle-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":17.2,"normalizedScore":16.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-hlenotools-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":8.7,"normalizedScore":6.5056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aagpqadiamond-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":79.2,"normalizedScore":78.8352,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aaomniscienceindex-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-48.1,"normalizedScore":30.6907,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-omniscienceaccuracy-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.2,"normalizedScore":25.7732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.9,"normalizedScore":19.421,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-26b-a4b-aaifbench-2026-08-01","modelSlug":"gemma-4-26b-a4b","modelName":"Gemma 4 26B A4B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":72.4,"normalizedScore":84.5133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 26B A4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-terminalbench2-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":54.4,"normalizedScore":33.274,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gertlabs-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":36.91,"normalizedScore":23.7954,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aaagenticindex-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.73,"normalizedScore":55.3919,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gdpvalaanormalized-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.8,"normalizedScore":52.5698,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gdpvalaa-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1215,"normalizedScore":67.3397,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-sweverified-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.4,"normalizedScore":70.1657,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-scicode-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":41.2,"normalizedScore":42.9003,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aascicode-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.6,"normalizedScore":78.7521,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aacodingindex-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.8,"normalizedScore":72.8905,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-lcr-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-critpt-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.9,"normalizedScore":15.1703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gpqa-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":87.2,"normalizedScore":88.1581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-gpqadiamond-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":87.2,"normalizedScore":88.1581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-hle-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":25.5,"normalizedScore":31.2281,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-omniscienceaccuracy-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.5,"normalizedScore":48.6254,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-omnisciencehallucinationrate-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73,"normalizedScore":28.9505,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-aaomniscienceindex-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-18.5,"normalizedScore":53.9246,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-preview-ifbench-2026-08-01","modelSlug":"hy3-preview","modelName":"Hy3 Preview","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":63.1,"normalizedScore":53.0043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3 Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aaagenticindex-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.17,"normalizedScore":1.6367,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-tau2bench-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":17.3,"normalizedScore":17.4571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-gdpvalaa-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":63,"normalizedScore":9.1873,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aacodingindex-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":11.14,"normalizedScore":5.5265,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aascicode-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":25.9,"normalizedScore":42.1585,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-lcr-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17,"normalizedScore":22.4571,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-critpt-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aammmupro-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":40.1,"normalizedScore":23.3677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-designarenawebsite-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":998,"normalizedScore":37.0432,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-mmlu-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":80.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-gpqa-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":50.3,"normalizedScore":35.5115,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aagpqadiamond-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":51.2,"normalizedScore":39.0625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aahle-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.9,"normalizedScore":1.5936,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aaomniscienceindex-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-56.4,"normalizedScore":24.1758,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.3,"normalizedScore":17.354,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.4,"normalizedScore":20.0241,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-ifeval-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":83.2,"normalizedScore":65.13,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-aaifbench-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":32,"normalizedScore":24.9263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-nano-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-4-1-nano","modelName":"GPT-4.1 nano","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":1.034,"normalizedScore":1.1618,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 nano; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-tau2airline-2026-08-01","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-tau2airline","benchmarkName":"τ²-Bench Airline Domain","benchmarkCategory":"agents","benchmarkOrganisation":"Victor Barres, Honghua Dong, Soham Ray, Xujie Si, Karthik Narasimhan","benchmarkVersion":"2025","score":56.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-livecodebenchv6-2026-08-01","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":65.7,"normalizedScore":53.9209,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-sweverified-2026-08-01","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":53.2,"normalizedScore":40.884,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-mmlupro-2026-08-01","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":68.1,"normalizedScore":69.4081,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-gpqa-2026-08-01","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":57.3,"normalizedScore":45.4986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-gpqadiamond-2026-08-01","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":57.3,"normalizedScore":45.4986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-zaya1-74b-preview-aime2026-2026-08-01","modelSlug":"zaya1-74b-preview","modelName":"ZAYA1-74B-Preview","providerId":"zyphra","providerName":"Zyphra","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":76.4,"normalizedScore":61.2113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant ZAYA1-74B-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-terminalbench2-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":50,"normalizedScore":25.4448,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-osworldverified-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"2025","score":61.4,"normalizedScore":48.6957,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-vitabench-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vitabench","benchmarkName":"VITA-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Meituan LongCat Team","benchmarkVersion":"2025","score":17,"normalizedScore":4.6296,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-gertlabs-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":48.51,"normalizedScore":48.3094,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-jobbench-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":27.7,"normalizedScore":41.5205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-sweverified-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":77.2,"normalizedScore":74.0331,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-arcagi2-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":13.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-designarenawebsite-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1215,"normalizedScore":73.0897,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-gpqa-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":83.4,"normalizedScore":82.7365,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-aime2025-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":87,"normalizedScore":82.3259,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-frontiermathv2tiers13-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":13.495,"normalizedScore":15.1629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-frontiermathv2tier4-2026-08-01","modelSlug":"claude-sonnet-4-5","modelName":"Claude Sonnet 4.5","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-tau2bench-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":47.1,"normalizedScore":47.5277,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-gertlabs-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":25.65,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-sweverified-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":54.6,"normalizedScore":42.8177,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aascicode-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.1,"normalizedScore":62.7319,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-lcr-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61,"normalizedScore":80.5812,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-critpt-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aammmupro-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":61.2,"normalizedScore":59.622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-designarenawebsite-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1063,"normalizedScore":47.8405,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mmlu-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":90.2,"normalizedScore":86.3248,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-gpqa-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":66.3,"normalizedScore":58.3393,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aagpqadiamond-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":66.6,"normalizedScore":60.9375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aahle-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aaomniscienceindex-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-36.2,"normalizedScore":40.0314,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.2,"normalizedScore":36.0825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":79.6,"normalizedScore":20.9891,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-ifeval-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":87.4,"normalizedScore":77.5414,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-aaifbench-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43,"normalizedScore":41.1504,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":5.517,"normalizedScore":6.1989,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-frontiermathv2tier4-2026-08-01","modelSlug":"gpt-4-1","modelName":"GPT-4.1","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaagenticindex-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.45,"normalizedScore":25.7865,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-tau2bench-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":59.9,"normalizedScore":60.444,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gdpvalaanormalized-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.5,"normalizedScore":22.7606,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gdpvalaa-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":811,"normalizedScore":46.946,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gertlabs-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":35.26,"normalizedScore":20.3085,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaenterpriseopsgym-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.3,"normalizedScore":10.9375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaitbench-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.3,"normalizedScore":62.6482,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-terminalbenchhard-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":36.4,"normalizedScore":30.4245,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaharveylab-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":47.2,"normalizedScore":41.2639,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aatau3banking-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aatau3banking","benchmarkName":"Artificial Analysis Tau3-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.1,"normalizedScore":3.6842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-swerebench-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"swe-rebench","benchmarkName":"SWE-Rebench","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-Rebench","benchmarkVersion":"2026","score":41.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-reactnativeevals-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":75.2,"normalizedScore":16.7331,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aacodingindex-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.43,"normalizedScore":51.1661,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aascicode-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.4,"normalizedScore":71.6695,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-lcr-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62,"normalizedScore":81.9022,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-critpt-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-mmmupro-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":76.9,"normalizedScore":44.8387,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aammmupro-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":73.4,"normalizedScore":80.5842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-gpqa-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":84.3,"normalizedScore":84.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-mmlupro-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":85.2,"normalizedScore":93.7393,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-hle-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":26.5,"normalizedScore":32.9825,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-hlenotools-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":19.5,"normalizedScore":26.5799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aagpqadiamond-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.7,"normalizedScore":88.0682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaomniscienceindex-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-45.4,"normalizedScore":32.81,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-omniscienceaccuracy-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.9,"normalizedScore":28.6942,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.6,"normalizedScore":18.5766,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaopennessindex-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-31b-aaifbench-2026-08-01","modelSlug":"gemma-4-31b","modelName":"Gemma 4 31B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":75.6,"normalizedScore":89.233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 31B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tiers13-2026-08-01","modelSlug":"qwen3-235b-2507-reasoning","modelName":"Qwen3 235B 2507 (Reasoning)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":8.481,"normalizedScore":9.5292,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-235b-2507-reasoning-frontiermathv2tier4-2026-08-01","modelSlug":"qwen3-235b-2507-reasoning","modelName":"Qwen3 235B 2507 (Reasoning)","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 235B 2507 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-tau2bench-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":36.3,"normalizedScore":36.6297,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aaagenticindex-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.93,"normalizedScore":13.9298,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-gdpvalaanormalized-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.6,"normalizedScore":11.1601,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-gdpvalaa-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":651,"normalizedScore":38.8693,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-livecodebenchv6-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":72,"normalizedScore":64.4772,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aascicode-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.2,"normalizedScore":62.9005,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aacodingindex-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.96,"normalizedScore":33.5406,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-bbh-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":53,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mrcrv2-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"mrcr-v2","benchmarkName":"MRCRv2","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2025","score":43.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-lcr-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.3,"normalizedScore":73.0515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-critpt-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mmmupro-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":69.1,"normalizedScore":19.6774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mathvision-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mathvision","benchmarkName":"MathVision","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":79.7,"normalizedScore":27,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-medxpertqamm-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":48.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aammmupro-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":69.7,"normalizedScore":74.2268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-gpqa-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":78.8,"normalizedScore":76.1735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-gpqadiamond-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":78.8,"normalizedScore":76.1735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mmlupro-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":77.2,"normalizedScore":82.3563,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-hlenotools-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":5.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-mmmlu-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":83.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aahle-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.8,"normalizedScore":23.3068,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aaomniscienceindex-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-51.9,"normalizedScore":27.708,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-omniscienceaccuracy-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16,"normalizedScore":21.9931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.8,"normalizedScore":19.5416,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aaifbench-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.5,"normalizedScore":86.1357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-12b-aime2026-2026-08-01","modelSlug":"gemma-4-12b","modelName":"Gemma 4 12B Unified","providerId":"google","providerName":"Google","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":77.5,"normalizedScore":63.0827,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 12B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-terminalbench2-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":47.1,"normalizedScore":20.2847,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-deepsearchqa-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":62.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-gertlabs-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":38.36,"normalizedScore":26.8597,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-livecodebenchpro-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":74.2,"normalizedScore":75.6312,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-sweverified-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":76.7,"normalizedScore":73.3425,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-swepro-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":51.8,"normalizedScore":23.7968,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-vibecodebench-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":4.064,"normalizedScore":5.7237,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-arcagi2-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"arc-agi-2","benchmarkName":"ARC-AGI-2","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2025","score":53.3,"normalizedScore":50.3169,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-arcagi3-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"arc-agi-3","benchmarkName":"ARC-AGI-3","benchmarkCategory":"reasoning","benchmarkOrganisation":"ARC Prize Foundation","benchmarkVersion":"2026","score":0.09,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-mmmupro-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":75.2,"normalizedScore":39.3548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-charxiv-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":60.9,"normalizedScore":20.098,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-erqa-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-erqa","benchmarkName":"ERQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":54.1,"normalizedScore":13.7363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-simplevqa-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-simplevqa","benchmarkName":"SimpleVQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":57.4,"normalizedScore":5.0781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-medxpertqamm-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-medxpertqamm","benchmarkName":"MedXpertQA Multimodal","benchmarkCategory":"multimodal","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":65.8,"normalizedScore":52.454,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-designarenawebsite-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1252,"normalizedScore":79.2359,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-gpqadiamond-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":88.5,"normalizedScore":90.0128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-hlenotools-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-hlenotools","benchmarkName":"Humanity's Last Exam without tools","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":31.6,"normalizedScore":49.0706,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-healthbenchhard-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"healthbench-hard","benchmarkName":"HealthBench Hard","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":20.3,"normalizedScore":19.6429,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-20-beta-medxpertqatext-2026-08-01","modelSlug":"grok-4-20","modelName":"Grok 4.20","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-medxpertqatext","benchmarkName":"MedXpertQA Text","benchmarkCategory":"knowledge","benchmarkOrganisation":"Meta AI","benchmarkVersion":"2026","score":50.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.20; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aaagenticindex-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.72,"normalizedScore":2.6368,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-tau2bench-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":52.9,"normalizedScore":53.3804,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.4,"normalizedScore":0.5874,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-gdpvalaa-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":508,"normalizedScore":31.6507,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-sweverified-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":23.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aacodingindex-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.21,"normalizedScore":18.3463,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aascicode-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.4,"normalizedScore":66.6105,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-lcr-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":42.3,"normalizedScore":55.8785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-critpt-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aammmupro-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":58.7,"normalizedScore":55.3265,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-designarenawebsite-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1022,"normalizedScore":41.0299,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-mmlu-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":87.5,"normalizedScore":63.2479,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-gpqa-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":64.2,"normalizedScore":55.3431,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aagpqadiamond-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":66.4,"normalizedScore":60.6534,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aahle-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aaomniscienceindex-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-50.1,"normalizedScore":29.1209,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.5,"normalizedScore":24.5704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82,"normalizedScore":18.0941,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-ifeval-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":88.5,"normalizedScore":80.792,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-aaifbench-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":38.3,"normalizedScore":34.2183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-1-mini-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-4-1-mini","modelName":"GPT-4.1 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.483,"normalizedScore":5.0371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4.1 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-tau2bench-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":14.9,"normalizedScore":15.0353,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aascicode-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.1,"normalizedScore":47.5548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-lcr-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.9,"normalizedScore":60.6341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-critpt-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aammmupro-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":65.5,"normalizedScore":67.0103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-designarenawebsite-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1140,"normalizedScore":60.6312,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aagpqadiamond-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68.3,"normalizedScore":63.3523,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aahle-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.1,"normalizedScore":3.9841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aaomniscienceindex-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-42,"normalizedScore":35.4788,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-omniscienceaccuracy-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.5,"normalizedScore":40.0344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.3,"normalizedScore":4.4632,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-aaifbench-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39,"normalizedScore":35.2507,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-frontiermathv2tiers13-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.844,"normalizedScore":5.4427,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-flash-frontiermathv2tier4-2026-08-01","modelSlug":"gemini-2-5-flash","modelName":"Gemini 2.5 Flash","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-flash-frontiermathv2tiers13-2026-08-01","modelSlug":"qwen3-5-flash","modelName":"Qwen3.5 Flash","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":6.207,"normalizedScore":6.9742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-5-flash-frontiermathv2tier4-2026-08-01","modelSlug":"qwen3-5-flash","modelName":"Qwen3.5 Flash","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3.5 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-tau2bench-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":76.9,"normalizedScore":77.5984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-vibecodebench-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":3.09,"normalizedScore":4.3519,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aascicode-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.1,"normalizedScore":54.3002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-lcr-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.3,"normalizedScore":34.7424,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-critpt-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aagpqadiamond-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":63.2,"normalizedScore":56.108,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aahle-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.2,"normalizedScore":4.1833,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aaomniscienceindex-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-31.6,"normalizedScore":43.6421,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-omniscienceaccuracy-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.8,"normalizedScore":30.2405,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-omnisciencehallucinationrate-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.1,"normalizedScore":37.2738,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-aaifbench-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.7,"normalizedScore":31.8584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-frontiermathv2tiers13-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":3.819,"normalizedScore":4.291,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-6-frontiermathv2tier4-2026-08-01","modelSlug":"glm-4-6","modelName":"GLM-4.6","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.128,"normalizedScore":2.5639,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.6; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-bigcodebench-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"bigcodebench","benchmarkName":"BigCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"BigCodeBench authors","benchmarkVersion":"2026","score":56.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-humaneval-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-humaneval","benchmarkName":"Evaluating Large Language Models Trained on Code","benchmarkCategory":"coding","benchmarkOrganisation":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, Alex Ray, Raul Puri, Gretchen Krueger, Michael Petrov, Heidy Khlaaf, Girish Sastry, Pamela Mishkin, Brooke Chan, Scott Gray, Nick Ryder, Mikhail Pavlov, Alethea Power, Lukasz Kaiser, Mohammad Bavarian, Clemens Winter, Philippe Tillet, Felipe Petroski Such, Dave Cummings, Matthias Plappert, Fotios Chantzis, Elizabeth Barnes, Ariel Herbert-Voss, William Hebgen Guss, Alex Nichol, Alex Paino, Nikolas Tezak, Jie Tang, Igor Babuschkin, Suchir Balaji, Shantanu Jain, William Saunders, Christopher Hesse, Andrew N. Carr, Jan Leike, Josh Achiam, Vedant Misra, Evan Morikawa, Alec Radford, Matthew Knight, Miles Brundage, Mira Murati, Katie Mayer, Peter Welinder, Bob McGrew, Dario Amodei, Sam McCandlish, Ilya Sutskever, Wojciech Zaremba","benchmarkVersion":"2021","score":69.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-bbh-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":86.9,"normalizedScore":98.2609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-drop-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-drop","benchmarkName":"Discrete Reasoning Over Paragraphs","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":88.6,"normalizedScore":99.5495,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-hellaswag-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hellaswag","benchmarkName":"HellaSwag","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":85.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-winogrande-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-winogrande","benchmarkName":"WinoGrande","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":79.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-cluewsc-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cluewsc","benchmarkName":"CLUEWSC","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":82.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-longbenchv2-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"longbench-v2","benchmarkName":"LongBench v2","benchmarkCategory":"long-context","benchmarkOrganisation":"THUDM","benchmarkVersion":"2025","score":44.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-agieval-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-agieval","benchmarkName":"AGIEval","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":82.6,"normalizedScore":96.9136,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmlu-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":88.7,"normalizedScore":73.5043,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmluredux-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":89.4,"normalizedScore":72.8711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmlupro-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":68.3,"normalizedScore":69.6927,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mmmlu-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmmlu","benchmarkName":"MMMLU","benchmarkCategory":"knowledge","benchmarkOrganisation":"OpenAI","benchmarkVersion":"2026","score":88.8,"normalizedScore":72,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-ceval-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-ceval","benchmarkName":"C-Eval","benchmarkCategory":"knowledge","benchmarkOrganisation":"C-Eval authors","benchmarkVersion":"2023","score":92.1,"normalizedScore":63.6364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-cmmlu-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmmlu","benchmarkName":"Chinese Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-multiloko-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-multiloko","benchmarkName":"MultiLoKo","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":42.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-simpleqa-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":30.1,"normalizedScore":20.1149,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-supergpqa-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":46.5,"normalizedScore":32.5077,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-factsparametric-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-factsparametric","benchmarkName":"FACTS Parametric","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":33.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-triviaqa-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-triviaqa","benchmarkName":"TriviaQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":82.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mgsm-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mgsm","benchmarkName":"Multilingual Grade School Math","benchmarkCategory":"knowledge","benchmarkOrganisation":"Freda Shi, Mirac Suzgun, Markus Freitag, Xuezhi Wang, Suraj Srivats, Soroush Vosoughi, Hyung Won Chung, Yi Tay, Sebastian Ruder, Denny Zhou, Dipanjan Das, Jason Wei","benchmarkVersion":"2022","score":85.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-gsm8k-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gsm8k","benchmarkName":"Grade School Math 8K","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":90.8,"normalizedScore":72.3077,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-mathbenchmark-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mathbenchmark","benchmarkName":"MATH","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":57.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-base-cmath-2026-08-01","modelSlug":"deepseek-v4-flash-base","modelName":"DeepSeek V4 Flash Base","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-cmath","benchmarkName":"CMath","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":93.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash Base; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-3-beta-frontiermathv2tiers13-2026-08-01","modelSlug":"grok-3-beta","modelName":"Grok 3 [Beta]","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":3.793,"normalizedScore":4.2618,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-3-beta-frontiermathv2tier4-2026-08-01","modelSlug":"grok-3-beta","modelName":"Grok 3 [Beta]","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 3 [Beta]; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-terminalbench2-2026-08-01","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":45.8,"normalizedScore":17.9715,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-sweverified-2026-08-01","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.6,"normalizedScore":70.442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-swemultilingual-2026-08-01","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":63.1,"normalizedScore":27.6119,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-m-1-swepro-2026-08-01","modelSlug":"laguna-m-1","modelName":"Laguna M.1","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":49.2,"normalizedScore":16.8449,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna M.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-bfclv4-2026-08-01","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":44.2,"normalizedScore":42.9313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-livecodebenchv6-2026-08-01","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":37.2,"normalizedScore":6.1662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-mmluredux-2026-08-01","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":78.1,"normalizedScore":30.2939,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-gpqa-2026-08-01","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":40.9,"normalizedScore":22.1002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-gpqadiamond-2026-08-01","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":40.9,"normalizedScore":22.1002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mellum2-12b-a2-5b-instruct-ifeval-2026-08-01","modelSlug":"mellum2-12b-a2-5b-instruct","modelName":"Mellum2-12B-A2.5B-Instruct","providerId":"jetbrains","providerName":"JetBrains","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":75.8,"normalizedScore":43.2624,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mellum2-12B-A2.5B-Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aaagenticindex-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.58,"normalizedScore":2.3823,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-tau2bench-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":22.8,"normalizedScore":23.0071,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-gdpvalaanormalized-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-gdpvalaa-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":231,"normalizedScore":17.6678,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-livecodebench-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"2024","score":37.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-sweverified-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":42,"normalizedScore":25.4144,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aacodingindex-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.04,"normalizedScore":22.3463,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aascicode-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.4,"normalizedScore":58.1788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-lcr-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29,"normalizedScore":38.3091,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-critpt-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-designarenawebsite-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1145,"normalizedScore":61.4618,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-gpqa-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":59.1,"normalizedScore":48.0668,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-mmlupro-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":75.9,"normalizedScore":80.5065,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aagpqadiamond-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":55.7,"normalizedScore":45.4545,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aahle-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.6,"normalizedScore":0.996,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aaomniscienceindex-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-41.3,"normalizedScore":36.0283,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-omniscienceaccuracy-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.4,"normalizedScore":38.1443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-omnisciencehallucinationrate-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.4,"normalizedScore":9.1677,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-ifeval-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":86.1,"normalizedScore":73.6998,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-aaifbench-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":34.8,"normalizedScore":29.056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-frontiermathv2tiers13-2026-08-01","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":1.724,"normalizedScore":1.9371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-sweverified-2026-08-01","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":49,"normalizedScore":35.0829,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-gpqa-2026-08-01","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":59.4,"normalizedScore":48.4948,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-frontiermathv2tiers13-2026-08-01","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":2.069,"normalizedScore":2.3247,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-5-sonnet-frontiermathv2tier4-2026-08-01","modelSlug":"claude-3-5-sonnet","modelName":"Claude 3.5 Sonnet","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3.5 Sonnet; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-tau2bench-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":25.1,"normalizedScore":25.328,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aascicode-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.3,"normalizedScore":54.6374,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-lcr-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-critpt-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-designarenawebsite-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":856,"normalizedScore":13.4551,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aagpqadiamond-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":54.3,"normalizedScore":43.4659,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aahle-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.3,"normalizedScore":0.3984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aaomniscienceindex-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.7,"normalizedScore":60.0471,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.7,"normalizedScore":28.3505,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.9,"normalizedScore":71.2907,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-aaifbench-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":34.3,"normalizedScore":28.3186,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-frontiermathv2tiers13-2026-08-01","modelSlug":"gpt-4o","modelName":"GPT-4o","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0.345,"normalizedScore":0.3876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aaagenticindex-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.1,"normalizedScore":1.5094,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-tau2bench-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":15.5,"normalizedScore":15.6408,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-gdpvalaanormalized-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-gdpvalaa-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":111,"normalizedScore":11.6103,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aacodingindex-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.17,"normalizedScore":1.3286,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aascicode-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":17,"normalizedScore":27.1501,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-lcr-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.8,"normalizedScore":34.0819,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-critpt-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aammmupro-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":52.9,"normalizedScore":45.3608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-designarenawebsite-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":775,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aagpqadiamond-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":58.7,"normalizedScore":49.7159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aahle-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.3,"normalizedScore":2.3904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aaomniscienceindex-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-52.4,"normalizedScore":27.3155,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-omniscienceaccuracy-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.6,"normalizedScore":19.5876,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-omnisciencehallucinationrate-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":78.3,"normalizedScore":22.5573,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-aaifbench-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.5,"normalizedScore":35.9882,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-scout-frontiermathv2tiers13-2026-08-01","modelSlug":"llama-4-scout","modelName":"Llama 4 Scout","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Scout; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aaagenticindex-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.31,"normalizedScore":1.8913,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-tau2bench-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":17.8,"normalizedScore":17.9617,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-gdpvalaanormalized-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-gdpvalaa-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":7,"normalizedScore":6.3604,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aacodingindex-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.28,"normalizedScore":12.7915,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aascicode-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.1,"normalizedScore":54.3002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-lcr-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46,"normalizedScore":60.7662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-critpt-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aammmupro-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":62.1,"normalizedScore":61.1684,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-designarenawebsite-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":896,"normalizedScore":20.0997,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aagpqadiamond-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":67.1,"normalizedScore":61.6477,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aahle-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.8,"normalizedScore":3.3865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aaomniscienceindex-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-41.8,"normalizedScore":35.6358,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-omniscienceaccuracy-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.3,"normalizedScore":36.2543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-omnisciencehallucinationrate-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.3,"normalizedScore":11.7008,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-aaifbench-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":43,"normalizedScore":41.1504,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-4-maverick-frontiermathv2tiers13-2026-08-01","modelSlug":"llama-4-maverick","modelName":"Llama 4 Maverick","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":0.69,"normalizedScore":0.7753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 4 Maverick; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-tau2bench-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86,"normalizedScore":86.781,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-gdpvalaanormalized-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.5,"normalizedScore":3.6711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-gdpvalaa-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":550,"normalizedScore":33.7708,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aaagenticindex-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":2.25,"normalizedScore":3.6007,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-scicode-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-scicode","benchmarkName":"Scientific Code Benchmark","benchmarkCategory":"coding","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2024","score":27,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aacodingindex-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.26,"normalizedScore":25.4841,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aascicode-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":27.1,"normalizedScore":44.1821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-lcr-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25,"normalizedScore":33.0251,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-critpt-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-gpqa-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":59,"normalizedScore":47.9241,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aagpqadiamond-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":59.3,"normalizedScore":50.5682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aahle-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.2,"normalizedScore":6.1753,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aaomniscienceindex-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-65.7,"normalizedScore":16.876,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-omniscienceaccuracy-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.4,"normalizedScore":20.9622,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-omnisciencehallucinationrate-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":95.8,"normalizedScore":1.4475,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-ifbench-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":57,"normalizedScore":39.9142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ling-2-6-flash-aaifbench-2026-08-01","modelSlug":"ling-2-6-flash","modelName":"Ling 2.6 Flash","providerId":"inclusionai","providerName":"InclusionAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":57.4,"normalizedScore":62.3894,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ling 2.6 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aaagenticindex-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.1,"normalizedScore":12.4204,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-tau2bench-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":54.1,"normalizedScore":54.5913,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gertlabs-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":42.01,"normalizedScore":34.5731,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gdpvalaanormalized-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.6,"normalizedScore":12.6285,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gdpvalaa-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":672,"normalizedScore":39.9293,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-sweverified-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":63.8,"normalizedScore":55.5249,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-vibecodebench-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":0.4,"normalizedScore":0.5634,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aacodingindex-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.25,"normalizedScore":36.7774,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aascicode-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42.8,"normalizedScore":70.6577,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-lcr-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66,"normalizedScore":87.1863,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-critpt-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.6,"normalizedScore":8.0495,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aammmupro-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.9,"normalizedScore":83.1615,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-designarenawebsite-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1192,"normalizedScore":69.2691,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-gpqa-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":83,"normalizedScore":82.1658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-hle-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":18.8,"normalizedScore":19.4737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aagpqadiamond-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.4,"normalizedScore":86.2216,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aaomniscienceindex-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-14.3,"normalizedScore":57.2214,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-omniscienceaccuracy-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39,"normalizedScore":61.512,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.4,"normalizedScore":11.5802,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-aaifbench-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":48.7,"normalizedScore":49.5575,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-frontiermathv2tiers13-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tiers13","benchmarkName":"FrontierMath v2 Tiers 1-3","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":14.138,"normalizedScore":15.8854,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-2-5-pro-frontiermathv2tier4-2026-08-01","modelSlug":"gemini-2-5-pro","modelName":"Gemini 2.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-frontiermathv2tier4","benchmarkName":"FrontierMath v2 Tier 4","benchmarkCategory":"mathematics","benchmarkOrganisation":"Epoch AI","benchmarkVersion":"2026","score":4.167,"normalizedScore":5.0205,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 2.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-terminalbench2-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":49.1,"normalizedScore":23.8434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-mcpatlas-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":64,"normalizedScore":58.8737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-toolathlon-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"toolathlon","benchmarkName":"Toolathlon","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":40.7,"normalizedScore":28.3368,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-claweval-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.8,"normalizedScore":73.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-gertlabs-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":54.35,"normalizedScore":60.6509,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-livecodebenchpass1cot-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-livecodebenchpass1cot","benchmarkName":"LiveCodeBench Pass@1 with Chain-of-Thought","benchmarkCategory":"coding","benchmarkOrganisation":"DeepSeek","benchmarkVersion":"2026","score":55.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-sweverified-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":73.7,"normalizedScore":69.1989,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-swepro-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":49.1,"normalizedScore":16.5775,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-swemultilingual-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":69.7,"normalizedScore":44.0299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-mrcr1m-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mrcr1m","benchmarkName":"MRCR 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":37.5,"normalizedScore":19.1564,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-corpusqa1m-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-corpusqa1m","benchmarkName":"CorpusQA 1M","benchmarkCategory":"reasoning","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":15.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-designarenawebsite-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1233,"normalizedScore":76.0797,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-mmlupro-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":83,"normalizedScore":90.609,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-simpleqa-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-simpleqa","benchmarkName":"Measuring Short-Form Factuality in Large Language Models","benchmarkCategory":"knowledge","benchmarkOrganisation":"Jason Wei, Najoung Kim, Hyung Won Chung, Yu-An Chung, Siddhartha Papay, Yifeng Lu, Hannaneh Hajishirzi, Luke Zettlemoyer","benchmarkVersion":"2024","score":23.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-chinesesimpleqa-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chinesesimpleqa","benchmarkName":"Chinese-SimpleQA","benchmarkCategory":"knowledge","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":71.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-gpqa-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":71.2,"normalizedScore":65.3303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-gpqadiamond-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":71.2,"normalizedScore":65.3303,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-hle-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2025","score":8.1,"normalizedScore":0.7018,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-hmmtfeb2026-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":40.8,"normalizedScore":21.0821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-imoanswerbench-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-imoanswerbench","benchmarkName":"IMOAnswerBench","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":41.9,"normalizedScore":12.0658,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-apex-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apex","benchmarkName":"Apex","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":1,"normalizedScore":1.3605,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v4-flash-apexshortlist-2026-08-01","modelSlug":"deepseek-v4-flash","modelName":"DeepSeek V4 Flash","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-apexshortlist","benchmarkName":"Apex Shortlist","benchmarkCategory":"mathematics","benchmarkOrganisation":"DeepSeek-AI","benchmarkVersion":"2026","score":9.3,"normalizedScore":0.1235,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V4 Flash; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-bfclv4-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":25.15,"normalizedScore":7.6339,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-livecodebenchpro-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-livecodebenchpro","benchmarkName":"LiveCodeBench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench Pro authors","benchmarkVersion":"2025","score":22.68,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-livecodebenchv6-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-livecodebenchv6","benchmarkName":"LiveCodeBench v6","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench maintainers","benchmarkVersion":"2026","score":33.52,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-bbh-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-bbh","benchmarkName":"BIG-Bench Hard","benchmarkCategory":"reasoning","benchmarkOrganisation":"Mirac Suzgun, Nathan Scales, Nathanael Schärli, Sebastian Gehrmann, Yi Tay, Hyung Won Chung, Aakanksha Chowdhery, Quoc V. Le, Ed H. Chi, Denny Zhou, Jason Wei","benchmarkVersion":"2022","score":71.89,"normalizedScore":54.7536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-mmlupro-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":48.85,"normalizedScore":42.0176,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-mmluredux-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-mmluredux","benchmarkName":"MMLU-Redux","benchmarkCategory":"knowledge","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":70.06,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-gpqadiamond-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":26.26,"normalizedScore":1.2127,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-supergpqa-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-supergpqa","benchmarkName":"SuperGPQA: Scaling LLM Evaluation Across 285 Graduate Disciplines","benchmarkCategory":"knowledge","benchmarkOrganisation":"Xiaoxuan Du, Yao Yao, Kexin Ma, Bowen Wang, Tianyu Zheng, Kaiyan Zhu, Yiming Zhang, Yutao Zhu, Jiawei Zhou, Jingren Zhou","benchmarkVersion":"2025","score":23.14,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-ifbench-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":46.67,"normalizedScore":17.7468,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-ifeval-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":80.41,"normalizedScore":56.8853,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-aime2025-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"aime-2025","benchmarkName":"AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2025","score":40.42,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-aime2026-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026","score":40.42,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-hmmtfeb2026-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-hmmtfeb2026","benchmarkName":"Harvard-MIT Mathematics Tournament February 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":25.76,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minicpm5-1b-math500-2026-08-01","modelSlug":"minicpm5-1b","modelName":"MiniCPM5-1B","providerId":"openbmb","providerName":"OpenBMB","benchmarkSlug":"benchlm-math500","benchmarkName":"MATH-500 Problem Set","benchmarkCategory":"mathematics","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Saurav Kadavath, Akul Arora, Steven Basart, Eric Tang, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2021","score":91.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniCPM5-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-terminalbench2-2026-08-01","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":43.1,"normalizedScore":13.1673,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-claweval-2026-08-01","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":63.1,"normalizedScore":80.4469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-sweverified-2026-08-01","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":69.4,"normalizedScore":63.2597,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-swepro-2026-08-01","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":42.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-swemultilingual-2026-08-01","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":52,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-ornith-1-0-9b-nl2repo-2026-08-01","modelSlug":"ornith-1-0-9b","modelName":"Ornith-1.0-9B","providerId":"deepreinforce-ai","providerName":"DeepReinforce AI","benchmarkSlug":"benchlm-nl2repo","benchmarkName":"NL2Repo","benchmarkCategory":"coding","benchmarkOrganisation":"MiniMax","benchmarkVersion":"2026","score":27.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Ornith-1.0-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-terminalbench2-2026-08-01","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"terminal-bench-2","benchmarkName":"Terminal-Bench 2.0","benchmarkCategory":"coding","benchmarkOrganisation":"Terminal-Bench contributors","benchmarkVersion":"2026","score":35.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-sweverified-2026-08-01","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":69.9,"normalizedScore":63.9503,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-swemultilingual-2026-08-01","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"benchlm-swemultilingual-benchmark-2","benchmarkName":"SWE-bench Multilingual","benchmarkCategory":"knowledge","benchmarkOrganisation":"SWE-bench team","benchmarkVersion":"2025","score":57.7,"normalizedScore":14.1791,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-laguna-xs-2-swepro-2026-08-01","modelSlug":"laguna-xs-2","modelName":"Laguna XS.2","providerId":"poolside","providerName":"Poolside","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"2025","score":46.3,"normalizedScore":9.0909,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Laguna XS.2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-bfclv4-2026-08-01","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":21.08,"normalizedScore":0.0926,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-mmmu-2026-08-01","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":32.67,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-realworldqa-2026-08-01","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-realworldqa","benchmarkName":"RealWorldQA","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":58.43,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-countbench-2026-08-01","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-countbench","benchmarkName":"CountBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"Qwen","benchmarkVersion":"2026","score":73.31,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-gpqa-2026-08-01","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":25.66,"normalizedScore":0.3567,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-mmlupro-2026-08-01","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":19.32,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-ifeval-2026-08-01","modelSlug":"lfm2-5-vl-450m","modelName":"LFM2.5-VL-450M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":61.16,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-tau2bench-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":85,"normalizedScore":85.7719,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaagenticindex-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.16,"normalizedScore":16.1666,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-gdpvalaanormalized-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.9,"normalizedScore":16.0059,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-gdpvalaa-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":718,"normalizedScore":42.2514,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-terminalbenchhard-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":25,"normalizedScore":3.5377,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aacodingindex-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":27.85,"normalizedScore":29.1449,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aascicode-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.8,"normalizedScore":62.226,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-lcr-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46,"normalizedScore":60.7662,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-critpt-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-mmmu-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-mmmu","benchmarkName":"Massive Multi-discipline Multimodal Understanding","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU authors","benchmarkVersion":"2024","score":75.1,"normalizedScore":79.5612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-mmmupro-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-mmmupro","benchmarkName":"Massive Multi-discipline Multimodal Understanding Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":63,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-charxiv-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"2024","score":52.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aammmupro-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":63.2,"normalizedScore":63.0584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aagpqadiamond-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.1,"normalizedScore":74.4318,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aahle-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":11.4,"normalizedScore":16.5339,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaomniscienceindex-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-4,"normalizedScore":65.3061,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-omniscienceaccuracy-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.9,"normalizedScore":9.7938,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-omnisciencehallucinationrate-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.1,"normalizedScore":100,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaopennessindex-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-command-a-plus-aaifbench-2026-08-01","modelSlug":"command-a-plus","modelName":"Command A+","providerId":"cohere","providerName":"Cohere","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.9,"normalizedScore":86.7257,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Command A+; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-bfclv4-2026-08-01","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-bfclv4","benchmarkName":"Berkeley Function Calling Leaderboard v4","benchmarkCategory":"agents","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":21.03,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-gpqa-2026-08-01","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":25.41,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-gpqadiamond-2026-08-01","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":25.41,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-mmlupro-2026-08-01","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":20.25,"normalizedScore":1.3233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-ifeval-2026-08-01","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifeval","benchmarkName":"Instruction-Following Eval","benchmarkCategory":"instruction-following","benchmarkOrganisation":"Jeffrey Zhou, Tianjian Lu, Swaroop Mishra, Siddhartha Brahma, Sujoy Basu, Yi Luan, Denny Zhou, Le Hou","benchmarkVersion":"2023","score":71.71,"normalizedScore":31.1761,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-230m-ifbench-2026-08-01","modelSlug":"lfm2-5-230m","modelName":"LFM2.5-230M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-ifbench","benchmarkName":"Instruction Following Benchmark","benchmarkCategory":"instruction-following","benchmarkOrganisation":"BenchLM registry","benchmarkVersion":"2025","score":38.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-230M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-researchclawbench-2026-08-01","modelSlug":"grok-4-1","modelName":"Grok 4.1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":13.5,"normalizedScore":12.6437,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aaagenticindex-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.73,"normalizedScore":55.3919,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-gdpvalaanormalized-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":35.8,"normalizedScore":52.5698,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-gdpvalaa-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1215,"normalizedScore":67.3397,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aascicode-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.6,"normalizedScore":78.7521,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aacodingindex-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":58.8,"normalizedScore":72.8905,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-lcr-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-critpt-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.9,"normalizedScore":15.1703,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-designarenawebsite-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1206,"normalizedScore":71.5947,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aagpqadiamond-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.7,"normalizedScore":93.75,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aahle-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":31.6,"normalizedScore":56.7729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-aaomniscienceindex-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-18.5,"normalizedScore":53.9246,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-omniscienceaccuracy-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.5,"normalizedScore":48.6254,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-hy3-omnisciencehallucinationrate-2026-08-01","modelSlug":"hy3","modelName":"Hy3","providerId":"tencent","providerName":"Tencent","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":73,"normalizedScore":28.9505,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Hy3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-reasoning-vibecodebench-2026-08-01","modelSlug":"glm-5-reasoning","modelName":"GLM-5 (Reasoning)","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":23.359,"normalizedScore":32.8986,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-reasoning-designarenawebsite-2026-08-01","modelSlug":"glm-5-reasoning","modelName":"GLM-5 (Reasoning)","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1274,"normalizedScore":82.8904,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5 (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-claweval-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":55.8,"normalizedScore":70.2514,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-tau2bench-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aascicode-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.6,"normalizedScore":72.0067,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-lcr-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.7,"normalizedScore":80.1849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-critpt-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-designarenawebsite-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1296,"normalizedScore":86.5449,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aagpqadiamond-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.7,"normalizedScore":86.6477,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aahle-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":25.4,"normalizedScore":44.4223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aaomniscienceindex-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-15.1,"normalizedScore":56.5934,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-omniscienceaccuracy-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29,"normalizedScore":44.3299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-omnisciencehallucinationrate-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":62.2,"normalizedScore":41.9783,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5-turbo-aaifbench-2026-08-01","modelSlug":"glm-5-turbo","modelName":"GLM-5-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.2,"normalizedScore":85.6932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-claweval-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":57.8,"normalizedScore":73.0447,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-tau2bench-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":95,"normalizedScore":95.8628,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-gertlabs-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":36.68,"normalizedScore":23.3094,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-researchclawbench-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-researchclawbench","benchmarkName":"ResearchClawBench","benchmarkCategory":"agents","benchmarkOrganisation":"InternScience","benchmarkVersion":"2026","score":15.3,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-sweverified-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":78,"normalizedScore":75.1381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aascicode-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42.5,"normalizedScore":70.1518,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-lcr-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.7,"normalizedScore":80.1849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-critpt-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aagpqadiamond-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":87,"normalizedScore":89.9148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aahle-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.3,"normalizedScore":50.1992,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aaomniscienceindex-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.9,"normalizedScore":72.292,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-omniscienceaccuracy-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.8,"normalizedScore":40.5498,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-omnisciencehallucinationrate-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.9,"normalizedScore":80.9409,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-pro-aaifbench-2026-08-01","modelSlug":"mimo-v2-pro","modelName":"MiMo-V2-Pro","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":68.8,"normalizedScore":79.2035,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aaagenticindex-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.71,"normalizedScore":46.263,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-tau2bench-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":84.8,"normalizedScore":85.5701,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28.9,"normalizedScore":42.4376,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-gdpvalaa-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1079,"normalizedScore":60.4745,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-jobbench-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":8.53,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-vibecodebench-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":20.088,"normalizedScore":28.2918,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aacodingindex-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":37.78,"normalizedScore":43.1802,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aascicode-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":42.9,"normalizedScore":70.8263,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-lcr-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.6,"normalizedScore":99.8679,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-critpt-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.7,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aammmupro-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.2,"normalizedScore":81.9588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-designarenawebsite-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1209,"normalizedScore":72.093,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aagpqadiamond-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.4,"normalizedScore":87.642,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aahle-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":26.5,"normalizedScore":46.6135,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-8.1,"normalizedScore":62.0879,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.7,"normalizedScore":64.433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82.1,"normalizedScore":17.9735,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aaifbench-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":73.1,"normalizedScore":85.5457,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-high-aamath500-2026-08-01","modelSlug":"gpt-5-high","modelName":"GPT-5 (high)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aamath500","benchmarkName":"Artificial Analysis MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (high); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-claweval-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":53.8,"normalizedScore":67.4581,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-tau2bench-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":98.5,"normalizedScore":99.3946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-gertlabs-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":30.76,"normalizedScore":10.7988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aascicode-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":43.5,"normalizedScore":71.8381,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-lcr-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":61,"normalizedScore":80.5812,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-critpt-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aammmupro-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.8,"normalizedScore":79.5533,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-designarenawebsite-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1255,"normalizedScore":79.7342,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aagpqadiamond-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":80.9,"normalizedScore":81.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aahle-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":15.8,"normalizedScore":25.2988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aaomniscienceindex-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-19,"normalizedScore":53.5322,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-omniscienceaccuracy-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.1,"normalizedScore":44.5017,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-omnisciencehallucinationrate-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.9,"normalizedScore":35.1025,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-5v-turbo-aaifbench-2026-08-01","modelSlug":"glm-5v-turbo","modelName":"GLM-5V-Turbo","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":61.1,"normalizedScore":67.8466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-5V-Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-tau2bench-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":93.3,"normalizedScore":94.1473,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-vibecodebench-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":1.2,"normalizedScore":1.6901,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aascicode-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":44.2,"normalizedScore":73.0185,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-lcr-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":68,"normalizedScore":89.8283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-critpt-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.9,"normalizedScore":8.9783,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aammmupro-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":63.3,"normalizedScore":63.2302,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aagpqadiamond-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":85.3,"normalizedScore":87.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aahle-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":17.6,"normalizedScore":28.8845,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aaomniscienceindex-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-28.7,"normalizedScore":45.9184,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-omniscienceaccuracy-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.3,"normalizedScore":37.9725,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-omnisciencehallucinationrate-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.4,"normalizedScore":29.6743,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-reasoning-aaifbench-2026-08-01","modelSlug":"grok-4-1-fast-reasoning","modelName":"Grok 4.1 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":52.7,"normalizedScore":55.4572,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-claweval-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":45.2,"normalizedScore":55.4469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-tau2bench-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":91.2,"normalizedScore":92.0283,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-sweverified-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":74.8,"normalizedScore":70.7182,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aascicode-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.7,"normalizedScore":60.371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-lcr-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.7,"normalizedScore":88.111,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-critpt-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aammmupro-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":69.9,"normalizedScore":74.5704,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aagpqadiamond-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":82.8,"normalizedScore":83.9489,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aahle-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":19.9,"normalizedScore":33.4661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aaomniscienceindex-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-17.4,"normalizedScore":54.7881,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-omniscienceaccuracy-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.7,"normalizedScore":26.6323,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-omnisciencehallucinationrate-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.4,"normalizedScore":63.4499,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mimo-v2-omni-aaifbench-2026-08-01","modelSlug":"mimo-v2-omni","modelName":"MiMo-V2-Omni","providerId":"xiaomi","providerName":"Xiaomi","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":53.5,"normalizedScore":56.6372,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiMo-V2-Omni; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-thinking-vibecodebench-2026-08-01","modelSlug":"deepseek-v3-2-thinking","modelName":"DeepSeek V3.2 (Thinking)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":5.108,"normalizedScore":7.1941,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-2-thinking-designarenawebsite-2026-08-01","modelSlug":"deepseek-v3-2-thinking","modelName":"DeepSeek V3.2 (Thinking)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1200,"normalizedScore":70.598,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.2 (Thinking); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-tau2bench-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":63.7,"normalizedScore":64.2785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-gertlabs-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":47.32,"normalizedScore":45.7946,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aascicode-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.6,"normalizedScore":48.398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-lcr-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22,"normalizedScore":29.0621,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-critpt-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aammmupro-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":48.4,"normalizedScore":37.6289,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aagpqadiamond-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":63.7,"normalizedScore":56.8182,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aahle-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5,"normalizedScore":3.7849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aaomniscienceindex-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-50.9,"normalizedScore":28.4929,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-omniscienceaccuracy-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17,"normalizedScore":23.7113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-omnisciencehallucinationrate-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.8,"normalizedScore":18.3353,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-1-fast-aaifbench-2026-08-01","modelSlug":"grok-4-1-fast","modelName":"Grok 4.1 Fast","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.5,"normalizedScore":31.5634,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4.1 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-tau2bench-2026-08-01","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":34.8,"normalizedScore":35.116,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aascicode-2026-08-01","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.7,"normalizedScore":60.371,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-lcr-2026-08-01","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45,"normalizedScore":59.4452,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-critpt-2026-08-01","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-designarenawebsite-2026-08-01","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1147,"normalizedScore":61.794,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aagpqadiamond-2026-08-01","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":73.5,"normalizedScore":70.7386,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aahle-2026-08-01","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.3,"normalizedScore":6.3745,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aaomniscienceindex-2026-08-01","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-41.1,"normalizedScore":36.1852,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-omniscienceaccuracy-2026-08-01","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.1,"normalizedScore":34.1924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-omnisciencehallucinationrate-2026-08-01","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.5,"normalizedScore":16.2847,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-v3-1-aaifbench-2026-08-01","modelSlug":"deepseek-v3-1","modelName":"DeepSeek V3.1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":37.8,"normalizedScore":33.4808,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek V3.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aaagenticindex-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.52,"normalizedScore":9.5472,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-tau2bench-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":24.6,"normalizedScore":24.8234,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-gdpvalaanormalized-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7,"normalizedScore":10.279,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-gdpvalaa-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":640,"normalizedScore":38.314,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aacodingindex-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.07,"normalizedScore":18.1484,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aascicode-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.2,"normalizedScore":59.5278,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-lcr-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.7,"normalizedScore":45.8388,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-critpt-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aammmupro-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":55.7,"normalizedScore":50.1718,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aagpqadiamond-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68,"normalizedScore":62.9261,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aahle-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.1,"normalizedScore":1.992,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aaomniscienceindex-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-39.4,"normalizedScore":37.5196,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-omniscienceaccuracy-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.1,"normalizedScore":35.9107,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-omnisciencehallucinationrate-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.7,"normalizedScore":16.0434,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-3-aaifbench-2026-08-01","modelSlug":"mistral-large-3","modelName":"Mistral Large 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.2,"normalizedScore":31.1209,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-designarenawebsite-2026-08-01","modelSlug":"glm-4-5","modelName":"GLM-4.5","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1195,"normalizedScore":69.7674,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-tau2bench-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":65.8,"normalizedScore":66.3976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-vibecodebench-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aascicode-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":44.2,"normalizedScore":73.0185,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-lcr-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":64.7,"normalizedScore":85.469,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-critpt-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":2.9,"normalizedScore":8.9783,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aammmupro-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":61.8,"normalizedScore":60.6529,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aagpqadiamond-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.7,"normalizedScore":86.6477,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aahle-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":17,"normalizedScore":27.6892,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aaomniscienceindex-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-28.4,"normalizedScore":46.1538,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-omniscienceaccuracy-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.6,"normalizedScore":33.3333,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-omnisciencehallucinationrate-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66,"normalizedScore":37.3945,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-4-fast-reasoning-aaifbench-2026-08-01","modelSlug":"grok-4-fast-reasoning","modelName":"Grok 4 Fast (Reasoning)","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":50.5,"normalizedScore":52.2124,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok 4 Fast (Reasoning); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-tau2bench-2026-08-01","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":36.5,"normalizedScore":36.8315,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aascicode-2026-08-01","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.3,"normalizedScore":66.4418,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-lcr-2026-08-01","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":54.7,"normalizedScore":72.2589,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-critpt-2026-08-01","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aagpqadiamond-2026-08-01","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":81.3,"normalizedScore":81.8182,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aahle-2026-08-01","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.9,"normalizedScore":23.506,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aaomniscienceindex-2026-08-01","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-27.1,"normalizedScore":47.1743,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-omniscienceaccuracy-2026-08-01","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31,"normalizedScore":47.7663,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-omnisciencehallucinationrate-2026-08-01","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":84,"normalizedScore":15.6815,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-aaifbench-2026-08-01","modelSlug":"deepseek-r1","modelName":"DeepSeek-R1","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.6,"normalizedScore":36.1357,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek-R1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-minimax-m2-5-vibecodebench-2026-08-01","modelSlug":"minimax-m2-5","modelName":"MiniMax M2.5","providerId":"minimax","providerName":"MiniMax","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":14.852,"normalizedScore":20.9174,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant MiniMax M2.5; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-preview-aacodingindex-2026-08-01","modelSlug":"o1-preview","modelName":"o1-preview","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.05,"normalizedScore":37.9081,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1-preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-tau2bench-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":90.1,"normalizedScore":90.9183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-gdpvalaanormalized-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.2,"normalizedScore":4.699,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-gdpvalaa-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":564,"normalizedScore":34.4775,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aaagenticindex-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.65,"normalizedScore":6.1466,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aascicode-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.1,"normalizedScore":59.3592,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aacodingindex-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.77,"normalizedScore":26.2049,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-lcr-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33,"normalizedScore":43.5931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-critpt-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-designarenawebsite-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1162,"normalizedScore":64.2857,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-mmlu-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-mmlu","benchmarkName":"Massive Multitask Language Understanding","benchmarkCategory":"knowledge","benchmarkOrganisation":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, Jacob Steinhardt","benchmarkVersion":"2020","score":87.2,"normalizedScore":60.6838,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-mmluproarcee-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":75.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-gpqadiamond-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":63.3,"normalizedScore":54.0591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aahle-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.7,"normalizedScore":23.1076,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aaomniscienceindex-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-44.2,"normalizedScore":33.752,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-omniscienceaccuracy-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.8,"normalizedScore":33.677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-omnisciencehallucinationrate-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.6,"normalizedScore":12.5452,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aaifbench-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":56.3,"normalizedScore":60.767,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-preview-aime2025arcee-2026-08-01","modelSlug":"trinity-large-preview","modelName":"Trinity-Large-Preview","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":24,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Preview; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-tau2bench-2026-08-01","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":46.5,"normalizedScore":46.9223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aascicode-2026-08-01","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":30.6,"normalizedScore":50.0843,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-lcr-2026-08-01","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":43.7,"normalizedScore":57.7279,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-critpt-2026-08-01","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-designarenawebsite-2026-08-01","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1171,"normalizedScore":65.7807,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aagpqadiamond-2026-08-01","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":73.3,"normalizedScore":70.4545,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aahle-2026-08-01","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.8,"normalizedScore":7.3705,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aaomniscienceindex-2026-08-01","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-62.5,"normalizedScore":19.3878,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-omniscienceaccuracy-2026-08-01","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.5,"normalizedScore":21.134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-omnisciencehallucinationrate-2026-08-01","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":92.3,"normalizedScore":5.6695,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-glm-4-5-air-aaifbench-2026-08-01","modelSlug":"glm-4-5-air","modelName":"GLM-4.5-Air","providerId":"zai","providerName":"Z.AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":37.6,"normalizedScore":33.1858,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GLM-4.5-Air; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-tau2bench-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":90.1,"normalizedScore":90.9183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gertlabs-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":32.55,"normalizedScore":14.5816,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gdpvalaanormalized-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.2,"normalizedScore":4.699,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gdpvalaa-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":564,"normalizedScore":34.4775,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aaagenticindex-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.65,"normalizedScore":6.1466,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-sweverifiedarcee-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-sweverifiedarcee","benchmarkName":"SWE-bench Verified (mini-swe-agent-v2)","benchmarkCategory":"coding","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":63.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aascicode-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.1,"normalizedScore":59.3592,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aacodingindex-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.77,"normalizedScore":26.2049,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-lcr-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33,"normalizedScore":43.5931,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-critpt-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-designarenawebsite-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1162,"normalizedScore":64.2857,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-gpqadiamond-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2023","score":76.3,"normalizedScore":72.6066,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-mmluproarcee-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-mmluproarcee","benchmarkName":"MMLU-Pro first-party comparison snapshot","benchmarkCategory":"knowledge","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":83.4,"normalizedScore":58.9928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aahle-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":14.7,"normalizedScore":23.1076,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aaomniscienceindex-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-44.2,"normalizedScore":33.752,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-omniscienceaccuracy-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.8,"normalizedScore":33.677,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-omnisciencehallucinationrate-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":86.6,"normalizedScore":12.5452,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aaifbench-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":56.3,"normalizedScore":60.767,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-trinity-large-thinking-aime2025arcee-2026-08-01","modelSlug":"trinity-large-thinking","modelName":"Trinity-Large-Thinking","providerId":"arcee-ai","providerName":"Arcee AI","benchmarkSlug":"benchlm-aime2025arcee","benchmarkName":"AIME25 first-party comparison snapshot","benchmarkCategory":"mathematics","benchmarkOrganisation":"Arcee AI","benchmarkVersion":"2026","score":96.3,"normalizedScore":95.3826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Trinity-Large-Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aaagenticindex-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.27,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-tau2bench-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":10.5,"normalizedScore":10.5954,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-gdpvalaanormalized-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-gdpvalaa-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":-119,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aacodingindex-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":10.06,"normalizedScore":4,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aascicode-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":21.2,"normalizedScore":34.2327,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-lcr-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.7,"normalizedScore":7.5297,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-critpt-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aammmupro-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":48,"normalizedScore":36.9416,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aagpqadiamond-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":42.8,"normalizedScore":27.1307,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aahle-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.7,"normalizedScore":3.1873,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aaomniscienceindex-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-65.9,"normalizedScore":16.719,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-omniscienceaccuracy-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":12.5,"normalizedScore":15.9794,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.5,"normalizedScore":9.047,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-3-27b-aaifbench-2026-08-01","modelSlug":"gemma-3-27b","modelName":"Gemma 3 27B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":31.8,"normalizedScore":24.6313,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 3 27B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aaagenticindex-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.7,"normalizedScore":8.056,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-tau2bench-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":41.2,"normalizedScore":41.5742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-gdpvalaanormalized-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.6,"normalizedScore":6.7548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-gdpvalaa-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":592,"normalizedScore":35.891,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aacodingindex-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":26.64,"normalizedScore":27.4346,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aascicode-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38,"normalizedScore":62.5632,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-lcr-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":44.7,"normalizedScore":59.0489,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-critpt-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aammmupro-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":56.8,"normalizedScore":52.0619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aagpqadiamond-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.9,"normalizedScore":75.5682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aahle-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":9.5,"normalizedScore":12.749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aaomniscienceindex-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-29.9,"normalizedScore":44.9765,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-omniscienceaccuracy-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.1,"normalizedScore":32.4742,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-omnisciencehallucinationrate-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.8,"normalizedScore":36.4294,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-small-4-aaifbench-2026-08-01","modelSlug":"mistral-small-4","modelName":"Mistral Small 4","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":48.2,"normalizedScore":48.8201,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Small 4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaagenticindex-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.17,"normalizedScore":23.4588,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-apexagentsaa-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":3.1,"normalizedScore":5.1724,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-tau2bench-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":65.8,"normalizedScore":66.3976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.1,"normalizedScore":22.1733,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-gdpvalaa-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":802,"normalizedScore":46.4917,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-gertlabs-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":29.61,"normalizedScore":8.3686,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaitbench-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaitbench","benchmarkName":"Artificial Analysis ITBench-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaenterpriseopsgym-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaenterpriseopsgym","benchmarkName":"Artificial Analysis EnterpriseOps-Gym","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":25.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaharveylab-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaharveylab","benchmarkName":"Artificial Analysis Harvey LAB-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-terminalbenchhard-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"terminal-bench-hard","benchmarkName":"Terminal-Bench Hard","benchmarkCategory":"agents","benchmarkOrganisation":"Laude Institute / Artificial Analysis","benchmarkVersion":"2026","score":23.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-reactnativeevals-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71.6,"normalizedScore":2.3904,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aacodingindex-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.44,"normalizedScore":32.8057,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aascicode-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.9,"normalizedScore":64.0809,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aalivecodebench-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aalivecodebench","benchmarkName":"Artificial Analysis LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":87.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-lcr-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":50.7,"normalizedScore":66.9749,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-critpt-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-designarenawebsite-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":993,"normalizedScore":36.2126,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aagpqadiamond-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":78.2,"normalizedScore":77.4148,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aahle-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":18.5,"normalizedScore":30.6773,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaomniscienceindex-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-50,"normalizedScore":29.1994,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.5,"normalizedScore":31.4433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.2,"normalizedScore":6.9964,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaopennessindex-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaopennessindex","benchmarkName":"Artificial Analysis Openness Index","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":11.2,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aammlupro-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaglobalmmlulite-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaifbench-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":69,"normalizedScore":79.4985,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-120b-aaaime2025-2026-08-01","modelSlug":"gpt-oss-120b","modelName":"GPT-OSS 120B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaaime2025","benchmarkName":"Artificial Analysis AIME 2025","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.4,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 120B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-tau2bench-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83,"normalizedScore":83.7538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-vibecodebench-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":22.168,"normalizedScore":31.2212,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aascicode-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.2,"normalizedScore":66.2732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-lcr-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.3,"normalizedScore":88.9036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-critpt-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.7,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aammmupro-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.5,"normalizedScore":79.0378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aagpqadiamond-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86,"normalizedScore":88.4943,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aahle-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.4,"normalizedScore":40.4382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-6,"normalizedScore":63.7363,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.2,"normalizedScore":61.8557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.4,"normalizedScore":27.2618,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-max-aaifbench-2026-08-01","modelSlug":"gpt-5-1-codex-max","modelName":"GPT-5.1-Codex-Max","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70,"normalizedScore":80.9735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex-Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-tau2bench-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":92.1,"normalizedScore":92.9364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-gertlabs-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":51.79,"normalizedScore":55.2409,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-jobbench-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":26,"normalizedScore":37.8384,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-vibecodebench-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":37.912,"normalizedScore":53.3949,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aascicode-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":54.6,"normalizedScore":90.5565,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-lcr-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":75.7,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-critpt-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":8.7,"normalizedScore":26.935,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aammmupro-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":76.3,"normalizedScore":85.567,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aagpqadiamond-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.9,"normalizedScore":94.0341,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aahle-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":33.5,"normalizedScore":60.5578,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-2.5,"normalizedScore":66.4835,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":40.7,"normalizedScore":64.433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.8,"normalizedScore":29.1918,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-2-codex-aaifbench-2026-08-01","modelSlug":"gpt-5-2-codex","modelName":"GPT-5.2 Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":77.6,"normalizedScore":92.1829,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.2-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-tau2bench-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":86.5,"normalizedScore":87.2856,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aascicode-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":41.1,"normalizedScore":67.7909,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-lcr-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":72.8,"normalizedScore":96.1691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-critpt-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aammmupro-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74.3,"normalizedScore":82.1306,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-designarenawebsite-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1209,"normalizedScore":72.093,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aagpqadiamond-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.2,"normalizedScore":85.9375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aahle-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.5,"normalizedScore":40.6375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.1,"normalizedScore":60.5181,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.9,"normalizedScore":61.3402,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.1,"normalizedScore":20.386,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aaifbench-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70.6,"normalizedScore":81.8584,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-medium-aamath500-2026-08-01","modelSlug":"gpt-5-medium","modelName":"GPT-5 (medium)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aamath500","benchmarkName":"Artificial Analysis MATH-500","benchmarkCategory":"mathematics","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":99.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 (medium); bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aacodingindex-2026-08-01","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.53,"normalizedScore":17.3852,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aascicode-2026-08-01","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":23.3,"normalizedScore":37.774,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aagpqadiamond-2026-08-01","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":48.9,"normalizedScore":35.7955,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-opus-aahle-2026-08-01","modelSlug":"claude-3-opus","modelName":"Claude 3 Opus","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Opus; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-tau2bench-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":40.9,"normalizedScore":41.2714,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-gdpvalaanormalized-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-gdpvalaa-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":492,"normalizedScore":30.843,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aaagenticindex-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.99,"normalizedScore":3.1278,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aascicode-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.6,"normalizedScore":48.398,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aacodingindex-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":14.37,"normalizedScore":10.0919,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-lcr-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":33.7,"normalizedScore":44.5178,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-critpt-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":2.7864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aagpqadiamond-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":75.7,"normalizedScore":73.8636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aahle-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":10.2,"normalizedScore":14.1434,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aaomniscienceindex-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-51.6,"normalizedScore":27.9435,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-omniscienceaccuracy-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.1,"normalizedScore":23.8832,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-omnisciencehallucinationrate-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":82.9,"normalizedScore":17.0084,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-nano-30b-aaifbench-2026-08-01","modelSlug":"nemotron-3-nano-30b","modelName":"Nemotron 3 Nano 30B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":71.1,"normalizedScore":82.5959,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Nano 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aaagenticindex-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.1,"normalizedScore":5.1464,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-apexagentsaa-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"apex-agents","benchmarkName":"APEX-Agents-AA","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / Mercor","benchmarkVersion":"2026","score":0.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-tau2bench-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":60.2,"normalizedScore":60.7467,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.4,"normalizedScore":4.9927,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-gdpvalaa-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":567,"normalizedScore":34.629,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-reactnativeevals-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":71,"normalizedScore":0,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aacodingindex-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.7,"normalizedScore":19.0389,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aascicode-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":34.4,"normalizedScore":56.4924,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-lcr-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.7,"normalizedScore":40.5548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-critpt-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.4,"normalizedScore":4.3344,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-designarenawebsite-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":877,"normalizedScore":16.9435,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aagpqadiamond-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":68.8,"normalizedScore":64.0625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aahle-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":9.8,"normalizedScore":13.3466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aaomniscienceindex-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-63.9,"normalizedScore":18.2889,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.5,"normalizedScore":21.134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94.1,"normalizedScore":3.4982,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-oss-20b-aaifbench-2026-08-01","modelSlug":"gpt-oss-20b","modelName":"GPT-OSS 20B","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":65.1,"normalizedScore":73.7463,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-OSS 20B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aaagenticindex-2026-08-01","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0.96,"normalizedScore":1.2548,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-gdpvalaanormalized-2026-08-01","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-gdpvalaa-2026-08-01","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":241,"normalizedScore":18.1726,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aascicode-2026-08-01","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":22.9,"normalizedScore":37.0995,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aacodingindex-2026-08-01","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":11.38,"normalizedScore":5.8657,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aammmupro-2026-08-01","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":41.5,"normalizedScore":25.7732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aagpqadiamond-2026-08-01","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":42.6,"normalizedScore":26.8466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aahle-2026-08-01","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4,"normalizedScore":1.7928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4o-mini-aaifbench-2026-08-01","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":31,"normalizedScore":23.4513,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4o mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-tau2bench-2026-08-01","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":30.7,"normalizedScore":30.9788,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aascicode-2026-08-01","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.2,"normalizedScore":47.7234,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-lcr-2026-08-01","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.3,"normalizedScore":7.0013,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-critpt-2026-08-01","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aagpqadiamond-2026-08-01","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":48.6,"normalizedScore":35.3693,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aahle-2026-08-01","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4,"normalizedScore":1.7928,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aaomniscienceindex-2026-08-01","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-34,"normalizedScore":41.7582,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-omniscienceaccuracy-2026-08-01","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":20.1,"normalizedScore":29.0378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-omnisciencehallucinationrate-2026-08-01","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.8,"normalizedScore":35.2232,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-large-2-aaifbench-2026-08-01","modelSlug":"mistral-large-2","modelName":"Mistral Large 2","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":31.2,"normalizedScore":23.7463,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Large 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-tau2bench-2026-08-01","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":19,"normalizedScore":19.1726,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aascicode-2026-08-01","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.9,"normalizedScore":48.9039,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-lcr-2026-08-01","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.3,"normalizedScore":32.1004,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-critpt-2026-08-01","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aagpqadiamond-2026-08-01","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":51.5,"normalizedScore":39.4886,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aahle-2026-08-01","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.2,"normalizedScore":2.1912,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aaomniscienceindex-2026-08-01","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-17.3,"normalizedScore":54.8666,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-omniscienceaccuracy-2026-08-01","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":22.3,"normalizedScore":32.8179,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-omnisciencehallucinationrate-2026-08-01","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":51,"normalizedScore":55.4885,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-llama-3-1-405b-aaifbench-2026-08-01","modelSlug":"llama-3-1-405b","modelName":"Llama 3.1 405B","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39,"normalizedScore":35.2507,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Llama 3.1 405B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen2-5-coder-32b-instruct-aascicode-2026-08-01","modelSlug":"qwen2-5-coder-32b-instruct","modelName":"Qwen2.5 Coder 32B Instruct","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":27.1,"normalizedScore":44.1821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen2-5-coder-32b-instruct-aagpqadiamond-2026-08-01","modelSlug":"qwen2-5-coder-32b-instruct","modelName":"Qwen2.5 Coder 32B Instruct","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":41.7,"normalizedScore":25.5682,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen2-5-coder-32b-instruct-aahle-2026-08-01","modelSlug":"qwen2-5-coder-32b-instruct","modelName":"Qwen2.5 Coder 32B Instruct","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.8,"normalizedScore":1.3944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen2.5 Coder 32B Instruct; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-3-super-100b-claweval-2026-08-01","modelSlug":"nemotron-3-super-100b","modelName":"Nemotron 3 Super 100B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-claweval","benchmarkName":"Claw-Eval","benchmarkCategory":"agents","benchmarkOrganisation":"Bowen Ye, Rang Li, Qibin Yang, Yuanxin Liu, Linli Yao, Hanglong Lv, Zhihui Xie, Chenxin An, Lei Li, Lingpeng Kong, Qi Liu, Zhifang Sui, Tong Yang","benchmarkVersion":"2026","score":5.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron 3 Super 100B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aacodingindex-2026-08-01","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.63,"normalizedScore":23.1802,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aascicode-2026-08-01","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":29.5,"normalizedScore":48.2293,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aammmupro-2026-08-01","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":55,"normalizedScore":48.9691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aagpqadiamond-2026-08-01","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":58.9,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-5-pro-aahle-2026-08-01","modelSlug":"gemini-1-5-pro","modelName":"Gemini 1.5 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.9,"normalizedScore":3.5857,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.5 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-tau2bench-2026-08-01","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aascicode-2026-08-01","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":26,"normalizedScore":42.3272,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-lcr-2026-08-01","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-critpt-2026-08-01","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aagpqadiamond-2026-08-01","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":57.5,"normalizedScore":48.0114,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aahle-2026-08-01","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.1,"normalizedScore":1.992,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aaomniscienceindex-2026-08-01","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-56.7,"normalizedScore":23.9403,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-omniscienceaccuracy-2026-08-01","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.2,"normalizedScore":17.1821,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-omnisciencehallucinationrate-2026-08-01","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.5,"normalizedScore":19.9035,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-phi-4-aaifbench-2026-08-01","modelSlug":"phi-4","modelName":"Phi-4","providerId":"microsoft","providerName":"Microsoft","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":23.5,"normalizedScore":12.3894,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Phi-4; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-mini-vibecodebench-2026-08-01","modelSlug":"gpt-5-mini","modelName":"GPT-5 mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":14.171,"normalizedScore":19.9583,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5 mini; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o3-pro-aagpqadiamond-2026-08-01","modelSlug":"o3-pro","modelName":"o3-pro","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":84.5,"normalizedScore":86.3636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o3-pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-tau2bench-2026-08-01","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":75.7,"normalizedScore":76.3875,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-sweverified-2026-08-01","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"2024","score":70.8,"normalizedScore":65.1934,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aascicode-2026-08-01","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":36.2,"normalizedScore":59.5278,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-lcr-2026-08-01","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":48.3,"normalizedScore":63.8045,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-critpt-2026-08-01","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aagpqadiamond-2026-08-01","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":72.7,"normalizedScore":69.6023,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aahle-2026-08-01","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7.5,"normalizedScore":8.7649,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aaomniscienceindex-2026-08-01","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-36,"normalizedScore":40.1884,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-omniscienceaccuracy-2026-08-01","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":23.8,"normalizedScore":35.3952,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-omnisciencehallucinationrate-2026-08-01","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":78.5,"normalizedScore":22.316,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-code-fast-1-aaifbench-2026-08-01","modelSlug":"grok-code-fast-1","modelName":"Grok Code Fast 1","providerId":"xai","providerName":"xAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":41.4,"normalizedScore":38.7906,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Code Fast 1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-turbo-aacodingindex-2026-08-01","modelSlug":"gpt-4-turbo","modelName":"GPT-4 Turbo","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21.49,"normalizedScore":20.1555,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-turbo-aascicode-2026-08-01","modelSlug":"gpt-4-turbo","modelName":"GPT-4 Turbo","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":31.9,"normalizedScore":52.2766,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-4-turbo-aahle-2026-08-01","modelSlug":"gpt-4-turbo","modelName":"GPT-4 Turbo","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.3,"normalizedScore":0.3984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-4 Turbo; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-tau2bench-2026-08-01","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":11.4,"normalizedScore":11.5035,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aascicode-2026-08-01","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":34.7,"normalizedScore":56.9983,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-lcr-2026-08-01","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.3,"normalizedScore":9.6433,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-critpt-2026-08-01","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aagpqadiamond-2026-08-01","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":72.8,"normalizedScore":69.7443,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aahle-2026-08-01","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":8.1,"normalizedScore":9.9602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aaomniscienceindex-2026-08-01","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-45.5,"normalizedScore":32.7316,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-omniscienceaccuracy-2026-08-01","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19.9,"normalizedScore":28.6942,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-omnisciencehallucinationrate-2026-08-01","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":81.7,"normalizedScore":18.456,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nemotron-ultra-253b-aaifbench-2026-08-01","modelSlug":"nemotron-ultra-253b","modelName":"Nemotron Ultra 253B","providerId":"nvidia","providerName":"NVIDIA","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":38.2,"normalizedScore":34.0708,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nemotron Ultra 253B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-0-pro-aascicode-2026-08-01","modelSlug":"gemini-1-0-pro","modelName":"Gemini 1.0 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":11.7,"normalizedScore":18.2125,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-0-pro-aagpqadiamond-2026-08-01","modelSlug":"gemini-1-0-pro","modelName":"Gemini 1.0 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":27.7,"normalizedScore":5.6818,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-1-0-pro-aahle-2026-08-01","modelSlug":"gemini-1-0-pro","modelName":"Gemini 1.0 Pro","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.6,"normalizedScore":2.988,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 1.0 Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-tau2bench-2026-08-01","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":21.1,"normalizedScore":21.2916,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aascicode-2026-08-01","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":18.6,"normalizedScore":29.8482,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-lcr-2026-08-01","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":21,"normalizedScore":27.7411,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-critpt-2026-08-01","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aammmupro-2026-08-01","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":30.8,"normalizedScore":7.3883,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aagpqadiamond-2026-08-01","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":37.4,"normalizedScore":19.4602,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aahle-2026-08-01","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.9,"normalizedScore":1.5936,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aaomniscienceindex-2026-08-01","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-47.6,"normalizedScore":31.0832,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-omniscienceaccuracy-2026-08-01","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.2,"normalizedScore":24.055,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-omnisciencehallucinationrate-2026-08-01","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":78.2,"normalizedScore":22.6779,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-3-haiku-aaifbench-2026-08-01","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36.1,"normalizedScore":30.9735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 3 Haiku; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-tau2bench-2026-08-01","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":71.4,"normalizedScore":72.0484,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aascicode-2026-08-01","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.9,"normalizedScore":67.4536,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-lcr-2026-08-01","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.3,"normalizedScore":87.5826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-critpt-2026-08-01","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aammmupro-2026-08-01","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":67.9,"normalizedScore":71.134,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aagpqadiamond-2026-08-01","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":80.9,"normalizedScore":81.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aahle-2026-08-01","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":11.9,"normalizedScore":17.5299,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-4-1-opus-thinking-aaifbench-2026-08-01","modelSlug":"claude-4-1-opus-thinking","modelName":"Claude 4.1 Opus Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":55.4,"normalizedScore":59.4395,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude 4.1 Opus Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-tau2bench-2026-08-01","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":14,"normalizedScore":14.1271,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aascicode-2026-08-01","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":20.8,"normalizedScore":33.5582,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-lcr-2026-08-01","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":19,"normalizedScore":25.0991,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-critpt-2026-08-01","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aammmupro-2026-08-01","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":44.3,"normalizedScore":30.5842,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aagpqadiamond-2026-08-01","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":49.9,"normalizedScore":37.2159,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aahle-2026-08-01","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.4,"normalizedScore":0.5976,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aaomniscienceindex-2026-08-01","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-47.6,"normalizedScore":31.0832,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-omniscienceaccuracy-2026-08-01","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17,"normalizedScore":23.7113,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-omnisciencehallucinationrate-2026-08-01","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.9,"normalizedScore":23.0398,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-nova-pro-aaifbench-2026-08-01","modelSlug":"nova-pro","modelName":"Nova Pro","providerId":"amazon","providerName":"Amazon","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":38.1,"normalizedScore":33.9233,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Nova Pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"supported","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-tau2bench-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":8.5,"normalizedScore":8.5772,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aascicode-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":3,"normalizedScore":3.5413,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-lcr-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-critpt-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractjsonvalidity-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractjsonvalidity","benchmarkName":"Liquid image-to-JSON extraction JSON validity","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":99.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractschemaf1-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractschemaf1","benchmarkName":"Liquid image-to-JSON extraction schema consistency F1","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":99.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-liquidextractvlmjudge-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractvlmjudge","benchmarkName":"Liquid image-to-JSON extraction VLM judge score","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":90.6,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aammmupro-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":26.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aagpqadiamond-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":28.9,"normalizedScore":7.3864,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aahle-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.1,"normalizedScore":3.9841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aaomniscienceindex-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-83.9,"normalizedScore":2.5903,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-omniscienceaccuracy-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.2,"normalizedScore":3.4364,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-omnisciencehallucinationrate-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94,"normalizedScore":3.6188,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-1-6b-extract-aaifbench-2026-08-01","modelSlug":"lfm2-5-vl-1-6b-extract","modelName":"LFM2.5-VL-1.6B-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":33.1,"normalizedScore":26.5487,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-1.6B-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-extract-liquidextractjsonvalidity-2026-08-01","modelSlug":"lfm2-5-vl-450m-extract","modelName":"LFM2.5-VL-450M-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractjsonvalidity","benchmarkName":"Liquid image-to-JSON extraction JSON validity","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":98.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-extract-liquidextractschemaf1-2026-08-01","modelSlug":"lfm2-5-vl-450m-extract","modelName":"LFM2.5-VL-450M-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractschemaf1","benchmarkName":"Liquid image-to-JSON extraction schema consistency F1","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":98.8,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-vl-450m-extract-liquidextractvlmjudge-2026-08-01","modelSlug":"lfm2-5-vl-450m-extract","modelName":"LFM2.5-VL-450M-Extract","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-liquidextractvlmjudge","benchmarkName":"Liquid image-to-JSON extraction VLM judge score","benchmarkCategory":"multimodal","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":84.5,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-VL-450M-Extract; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-kimiclaw247-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-kimiclaw247","benchmarkName":"Kimi Claw 24/7 Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":46.9,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-mcpatlas-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"2026","score":76,"normalizedScore":79.3515,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-mcpmarkverified-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mcpmarkverified","benchmarkName":"MCPMark-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"MCPMark","benchmarkVersion":"2026","score":81.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aaagenticindex-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":29.59,"normalizedScore":53.3188,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-tau2bench-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":90.1,"normalizedScore":90.9183,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-gdpvalaanormalized-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":34.3,"normalizedScore":50.3671,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-gdpvalaa-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":1186,"normalizedScore":65.8758,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-kimicodebenchv2-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"kimi-code-bench-v2","benchmarkName":"Kimi Code Bench v2","benchmarkCategory":"coding","benchmarkOrganisation":"Moonshot AI","benchmarkVersion":"2026","score":62,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-programbench-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-programbench","benchmarkName":"ProgramBench: Can Language Models Rebuild Programs From Scratch?","benchmarkCategory":"coding","benchmarkOrganisation":"John Yang, Kilian Lieret, Jeffrey Ma, Parth Thakkar, Dmitrii Pedchenko, Sten Sootla, Emily McMilin, Pengcheng Yin, Rui Hou, Gabriel Synnaeve, Diyi Yang, Ofir Press","benchmarkVersion":"2026","score":53.6,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-mlsbenchlite-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-mlsbenchlite","benchmarkName":"MLS-Bench Lite","benchmarkCategory":"coding","benchmarkOrganisation":"MLS-Bench","benchmarkVersion":"2026","score":35.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aacodingindex-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.76,"normalizedScore":75.6608,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aascicode-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":47.5,"normalizedScore":78.5835,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-lcr-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":66.3,"normalizedScore":87.5826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-critpt-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":10,"normalizedScore":30.9598,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-designarenawebsite-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1300,"normalizedScore":87.2093,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aagpqadiamond-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":89.6,"normalizedScore":93.608,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aahle-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":32.8,"normalizedScore":59.1633,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aaomniscienceindex-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-10.7,"normalizedScore":60.0471,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-omniscienceaccuracy-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":38.6,"normalizedScore":60.8247,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-omnisciencehallucinationrate-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":80.3,"normalizedScore":20.1448,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-kimi-k2-7-code-aaifbench-2026-08-01","modelSlug":"kimi-k2-7-code","modelName":"Kimi K2.7 Code","providerId":"moonshot","providerName":"Moonshot AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":63.1,"normalizedScore":70.7965,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Kimi K2.7 Code; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-tau2bench-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":83,"normalizedScore":83.7538,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-gertlabs-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":49.68,"normalizedScore":50.7819,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-jobbench-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":26.2,"normalizedScore":38.2716,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-vibecodebench-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":13.115,"normalizedScore":18.4711,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aascicode-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":40.2,"normalizedScore":66.2732,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-lcr-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":67.3,"normalizedScore":88.9036,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-critpt-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":5.7,"normalizedScore":17.6471,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aammmupro-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":72.5,"normalizedScore":79.0378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-designarenawebsite-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1186,"normalizedScore":68.2724,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aagpqadiamond-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86,"normalizedScore":88.4943,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aahle-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":23.4,"normalizedScore":40.4382,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aaomniscienceindex-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-6,"normalizedScore":63.7363,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-omniscienceaccuracy-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":39.2,"normalizedScore":61.8557,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-omnisciencehallucinationrate-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74.4,"normalizedScore":27.2618,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gpt-5-1-codex-aaifbench-2026-08-01","modelSlug":"gpt-5-1-codex","modelName":"GPT-5.1-Codex","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":70,"normalizedScore":80.9735,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant GPT-5.1-Codex; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-cyber-cybergym-2026-08-01","modelSlug":"sakana-fugu-cyber","modelName":"Fugu Cyber","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":86.9,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sakana-fugu-cyber-ctirealm-2026-08-01","modelSlug":"sakana-fugu-cyber","modelName":"Fugu Cyber","providerId":"sakana-ai","providerName":"Sakana AI","benchmarkSlug":"benchlm-ctirealm","benchmarkName":"CTI-REALM","benchmarkCategory":"agents","benchmarkOrganisation":"Sakana AI","benchmarkVersion":"2026","score":72.1,"normalizedScore":50,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Fugu Cyber; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-1-35b-a3b-androidworld-2026-08-01","modelSlug":"holo3-1-35b-a3b","modelName":"Holo3.1-35B-A3B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":79.3,"normalizedScore":84.1121,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3.1-35B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-1-4b-androidworld-2026-08-01","modelSlug":"holo3-1-4b","modelName":"Holo3.1-4B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":71,"normalizedScore":6.5421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3.1-4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo3-1-9b-androidworld-2026-08-01","modelSlug":"holo3-1-9b","modelName":"Holo3.1-9B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"benchlm-androidworld","benchmarkName":"AndroidWorld","benchmarkCategory":"agents","benchmarkOrganisation":"Z.AI","benchmarkVersion":"2026","score":71,"normalizedScore":6.5421,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo3.1-9B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-tau2bench-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74.3,"normalizedScore":74.9748,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-gertlabs-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":43.74,"normalizedScore":38.2291,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-vibecodebench-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":3.506,"normalizedScore":4.9378,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aascicode-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":38.3,"normalizedScore":63.0691,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-lcr-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":46.7,"normalizedScore":61.6909,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-critpt-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-designarenawebsite-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1144,"normalizedScore":61.2957,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aagpqadiamond-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":76.4,"normalizedScore":74.858,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aahle-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":11.1,"normalizedScore":15.9363,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aaomniscienceindex-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-43.1,"normalizedScore":34.6154,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-omniscienceaccuracy-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":24.4,"normalizedScore":36.4261,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-omnisciencehallucinationrate-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.4,"normalizedScore":9.1677,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-qwen3-max-aaifbench-2026-08-01","modelSlug":"qwen3-max","modelName":"Qwen3 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":44.1,"normalizedScore":42.7729,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Qwen3 Max; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-composer-2-fast-reactnativeevals-2026-08-01","modelSlug":"composer-2-fast","modelName":"Composer 2 Fast","providerId":"cursor","providerName":"Cursor","benchmarkSlug":"benchlm-reactnativeevals","benchmarkName":"React Native Evals","benchmarkCategory":"coding","benchmarkOrganisation":"Callstack","benchmarkVersion":"2026","score":94.9,"normalizedScore":95.2191,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Composer 2 Fast; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-tau2bench-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":89.5,"normalizedScore":90.3128,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-vibecodebench-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":20.63,"normalizedScore":29.0551,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aascicode-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":49.5,"normalizedScore":81.9562,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-lcr-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":74,"normalizedScore":97.7543,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-critpt-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":4.6,"normalizedScore":14.2415,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aammmupro-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":74,"normalizedScore":81.6151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-designarenawebsite-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1272,"normalizedScore":82.5581,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aagpqadiamond-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":86.6,"normalizedScore":89.3466,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aahle-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":28.4,"normalizedScore":50.3984,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aaomniscienceindex-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":13.3,"normalizedScore":78.8854,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-omniscienceaccuracy-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":45.7,"normalizedScore":73.0241,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-omnisciencehallucinationrate-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":59.8,"normalizedScore":44.8733,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aammlupro-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aammlupro","benchmarkName":"Artificial Analysis MMLU-Pro","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.5,"normalizedScore":96.6667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aaglobalmmlulite-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-aaglobalmmlulite","benchmarkName":"Artificial Analysis Global-MMLU-Lite","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.3,"normalizedScore":81.7308,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-opus-4-5-thinking-aaifbench-2026-08-01","modelSlug":"claude-opus-4-5-thinking","modelName":"Claude Opus 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":58,"normalizedScore":63.2743,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Opus 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-thinking-vibecodebench-2026-08-01","modelSlug":"claude-haiku-4-5-thinking","modelName":"Claude Haiku 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":11.393,"normalizedScore":16.0458,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-haiku-4-5-thinking-designarenawebsite-2026-08-01","modelSlug":"claude-haiku-4-5-thinking","modelName":"Claude Haiku 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1147,"normalizedScore":61.794,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Haiku 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-thinking-vibecodebench-2026-08-01","modelSlug":"claude-sonnet-4-5-thinking","modelName":"Claude Sonnet 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-vibecodebench","benchmarkName":"Vibe Code Bench v1.1","benchmarkCategory":"coding","benchmarkOrganisation":"Vals AI","benchmarkVersion":"2026","score":22.621,"normalizedScore":31.8592,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-claude-sonnet-4-5-thinking-designarenawebsite-2026-08-01","modelSlug":"claude-sonnet-4-5-thinking","modelName":"Claude Sonnet 4.5 Thinking","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1215,"normalizedScore":73.0897,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Claude Sonnet 4.5 Thinking; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-235b-a22b-screenspotpro-2026-08-01","modelSlug":"holo2-235b-a22b","modelName":"Holo2-235B-A22B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":70.6,"normalizedScore":59.0047,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-235B-A22B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-30b-a3b-screenspotpro-2026-08-01","modelSlug":"holo2-30b-a3b","modelName":"Holo2-30B-A3B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":66.1,"normalizedScore":48.3412,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-30B-A3B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-4b-screenspotpro-2026-08-01","modelSlug":"holo2-4b","modelName":"Holo2-4B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":57.2,"normalizedScore":27.2512,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-holo2-8b-screenspotpro-2026-08-01","modelSlug":"holo2-8b","modelName":"Holo2-8B","providerId":"h-company","providerName":"H Company","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"2025","score":58.9,"normalizedScore":31.2796,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Holo2-8B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemini-3-5-flash-cyber-cybergym-2026-08-01","modelSlug":"gemini-3-5-flash-cyber","modelName":"Gemini 3.5 Flash Cyber","providerId":"google","providerName":"Google","benchmarkSlug":"cybergym","benchmarkName":"CyberGym","benchmarkCategory":"agents","benchmarkOrganisation":"UC Berkeley SunBlaze","benchmarkVersion":"2026","score":83.2,"normalizedScore":91.5332,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemini 3.5 Flash Cyber; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-colbert-350m-nanobeirmultilingual-2026-08-01","modelSlug":"lfm2-5-colbert-350m","modelName":"LFM2.5-ColBERT-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-nanobeirmultilingual","benchmarkName":"NanoBEIR Multilingual Extended","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":60.5,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-colbert-350m-mkqa11-2026-08-01","modelSlug":"lfm2-5-colbert-350m","modelName":"LFM2.5-ColBERT-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mkqa11","benchmarkName":"MKQA-11 multilingual retrieval","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":69.4,"normalizedScore":100,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-ColBERT-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-embedding-350m-nanobeirmultilingual-2026-08-01","modelSlug":"lfm2-5-embedding-350m","modelName":"LFM2.5-Embedding-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-nanobeirmultilingual","benchmarkName":"NanoBEIR Multilingual Extended","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":57.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-lfm2-5-embedding-350m-mkqa11-2026-08-01","modelSlug":"lfm2-5-embedding-350m","modelName":"LFM2.5-Embedding-350M","providerId":"liquidai","providerName":"LiquidAI","benchmarkSlug":"benchlm-mkqa11","benchmarkName":"MKQA-11 multilingual retrieval","benchmarkCategory":"knowledge","benchmarkOrganisation":"Liquid AI","benchmarkVersion":"2026","score":69.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant LFM2.5-Embedding-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-grok-build-0-1-gertlabs-2026-08-01","modelSlug":"grok-build-0-1","modelName":"Grok Build 0.1","providerId":"xai","providerName":"xAI","benchmarkSlug":"benchlm-gertlabs","benchmarkName":"Gert Labs Composite Game Benchmark","benchmarkCategory":"agents","benchmarkOrganisation":"Gert Labs","benchmarkVersion":"2026","score":49.15,"normalizedScore":49.6619,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Grok Build 0.1; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-tau2bench-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":20.8,"normalizedScore":20.9889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aaagenticindex-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.51,"normalizedScore":2.255,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-gdpvalaanormalized-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-gdpvalaa-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":88,"normalizedScore":10.4493,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aascicode-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":20.9,"normalizedScore":33.7268,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aacodingindex-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":7.23,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-lcr-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15,"normalizedScore":19.8151,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-critpt-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aammmupro-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":44.6,"normalizedScore":31.0997,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-gpqa-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":43.4,"normalizedScore":25.667,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-mmlupro-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":60,"normalizedScore":57.8828,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aagpqadiamond-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":40.5,"normalizedScore":23.8636,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aahle-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.8,"normalizedScore":3.3865,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aaomniscienceindex-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-24,"normalizedScore":49.6075,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-omniscienceaccuracy-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.7,"normalizedScore":6.0137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.9,"normalizedScore":77.3221,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e2b-aaifbench-2026-08-01","modelSlug":"gemma-4-e2b","modelName":"Gemma 4 E2B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":36,"normalizedScore":30.826,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-tau2bench-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":20.8,"normalizedScore":20.9889,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aaagenticindex-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":1.79,"normalizedScore":2.7641,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-gdpvalaanormalized-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-gdpvalaa-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":231,"normalizedScore":17.6678,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aascicode-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":24.4,"normalizedScore":39.629,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aacodingindex-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.39,"normalizedScore":3.053,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-lcr-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":30.7,"normalizedScore":40.5548,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-critpt-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.6,"normalizedScore":1.8576,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aammmupro-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":51.4,"normalizedScore":42.7835,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-gpqa-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":58.6,"normalizedScore":47.3534,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-mmlupro-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-mmlupro","benchmarkName":"Massive Multitask Language Understanding Professional","benchmarkCategory":"knowledge","benchmarkOrganisation":"Yubo Wang, Xueguang Ma, Ge Zhang, Yuansheng Ni, Abhranil Chandra, Shiguang Guo, Weiming Ren, Aaran Arulraj, Xuan He, Ziyan Jiang, Tianle Li, Max Ku, Kai Wang, Alex Zhuang, Rongqi Fan, Xiang Yue, Wenhu Chen","benchmarkVersion":"2024","score":69.4,"normalizedScore":71.2578,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aagpqadiamond-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":52.2,"normalizedScore":40.483,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aahle-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.7,"normalizedScore":1.1952,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aaomniscienceindex-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-20,"normalizedScore":52.7473,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-omniscienceaccuracy-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8.6,"normalizedScore":9.2784,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-omnisciencehallucinationrate-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":31.3,"normalizedScore":79.2521,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-gemma-4-e4b-aaifbench-2026-08-01","modelSlug":"gemma-4-e4b","modelName":"Gemma 4 E4B","providerId":"google","providerName":"Google","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":40.6,"normalizedScore":37.6106,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Gemma 4 E4B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-tau2bench-2026-08-01","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":46.8,"normalizedScore":47.225,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aascicode-2026-08-01","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":26.4,"normalizedScore":43.0017,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-lcr-2026-08-01","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-critpt-2026-08-01","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aagpqadiamond-2026-08-01","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":73.8,"normalizedScore":71.1648,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aahle-2026-08-01","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":10.1,"normalizedScore":13.9442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aaomniscienceindex-2026-08-01","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-59.5,"normalizedScore":21.7425,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-omniscienceaccuracy-2026-08-01","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":17.6,"normalizedScore":24.7423,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-omnisciencehallucinationrate-2026-08-01","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.5,"normalizedScore":4.222,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-105b-aaifbench-2026-08-01","modelSlug":"sarvam-105b","modelName":"Sarvam 105B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":34.4,"normalizedScore":28.4661,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 105B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-tau2bench-2026-08-01","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":34.5,"normalizedScore":34.8133,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aascicode-2026-08-01","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":19.2,"normalizedScore":30.86,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-lcr-2026-08-01","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-critpt-2026-08-01","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0.3,"normalizedScore":0.9288,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aagpqadiamond-2026-08-01","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":63.3,"normalizedScore":56.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aahle-2026-08-01","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":7,"normalizedScore":7.7689,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aaomniscienceindex-2026-08-01","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-72,"normalizedScore":11.9309,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-omniscienceaccuracy-2026-08-01","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":12.7,"normalizedScore":16.323,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-omnisciencehallucinationrate-2026-08-01","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":97,"normalizedScore":0,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-sarvam-30b-aaifbench-2026-08-01","modelSlug":"sarvam-30b","modelName":"Sarvam 30B","providerId":"sarvam","providerName":"Sarvam","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":26.5,"normalizedScore":16.8142,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Sarvam 30B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-tau2bench-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":24.3,"normalizedScore":24.5207,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aascicode-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":33.1,"normalizedScore":54.3002,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-lcr-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":28,"normalizedScore":36.9881,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-critpt-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aammmupro-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2026","score":53,"normalizedScore":45.5326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-designarenawebsite-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-designarenawebsite","benchmarkName":"Design Arena Website Elo","benchmarkCategory":"multimodal","benchmarkOrganisation":"Design Arena","benchmarkVersion":"2026","score":1104,"normalizedScore":54.6512,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aagpqadiamond-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":57.8,"normalizedScore":48.4375,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aahle-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":4.3,"normalizedScore":2.3904,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aaomniscienceindex-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-31.5,"normalizedScore":43.7206,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-omniscienceaccuracy-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":18.3,"normalizedScore":25.945,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-omnisciencehallucinationrate-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":60.9,"normalizedScore":43.5464,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-mistral-medium-3-aaifbench-2026-08-01","modelSlug":"mistral-medium-3","modelName":"Mistral Medium 3","providerId":"mistral","providerName":"Mistral AI","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":39.3,"normalizedScore":35.6932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Mistral Medium 3; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-tau2bench-2026-08-01","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":22.8,"normalizedScore":23.0071,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aascicode-2026-08-01","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":8.7,"normalizedScore":13.1535,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-lcr-2026-08-01","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4,"normalizedScore":5.284,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-critpt-2026-08-01","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aagpqadiamond-2026-08-01","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":28.1,"normalizedScore":6.25,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aahle-2026-08-01","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.1,"normalizedScore":3.9841,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aaomniscienceindex-2026-08-01","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-81.8,"normalizedScore":4.2386,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-omniscienceaccuracy-2026-08-01","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.1,"normalizedScore":4.9828,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-omnisciencehallucinationrate-2026-08-01","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":93.5,"normalizedScore":4.222,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-1b-aaifbench-2026-08-01","modelSlug":"granite-4-0-1b","modelName":"Granite-4.0-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":20.5,"normalizedScore":7.9646,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-tau2bench-2026-08-01","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":13.2,"normalizedScore":13.3199,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aascicode-2026-08-01","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":0.9,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-lcr-2026-08-01","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-critpt-2026-08-01","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aagpqadiamond-2026-08-01","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":23.7,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aahle-2026-08-01","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.7,"normalizedScore":5.1793,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aaomniscienceindex-2026-08-01","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-72.1,"normalizedScore":11.8524,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-omniscienceaccuracy-2026-08-01","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.2,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-omnisciencehallucinationrate-2026-08-01","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":77.8,"normalizedScore":23.1604,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-350m-aaifbench-2026-08-01","modelSlug":"granite-4-0-350m","modelName":"Granite-4.0-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":15.1,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-tau2bench-2026-08-01","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":19.6,"normalizedScore":19.778,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aascicode-2026-08-01","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":8.2,"normalizedScore":12.3103,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-lcr-2026-08-01","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":6.3,"normalizedScore":8.3223,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-critpt-2026-08-01","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aagpqadiamond-2026-08-01","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":24.6,"normalizedScore":1.2784,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aahle-2026-08-01","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5,"normalizedScore":3.7849,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aaomniscienceindex-2026-08-01","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-73.6,"normalizedScore":10.675,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-omniscienceaccuracy-2026-08-01","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":5.3,"normalizedScore":3.6082,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-omnisciencehallucinationrate-2026-08-01","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":83.4,"normalizedScore":16.4053,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-1b-aaifbench-2026-08-01","modelSlug":"granite-4-0-h-1b","modelName":"Granite-4.0-H-1B","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":25.1,"normalizedScore":14.7493,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-1B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-tau2bench-2026-08-01","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":14.6,"normalizedScore":14.7326,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aascicode-2026-08-01","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":1.7,"normalizedScore":1.3491,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-lcr-2026-08-01","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-critpt-2026-08-01","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aagpqadiamond-2026-08-01","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":29,"normalizedScore":7.5284,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aahle-2026-08-01","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":6.4,"normalizedScore":6.5737,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aaomniscienceindex-2026-08-01","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-87.2,"normalizedScore":0,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-omniscienceaccuracy-2026-08-01","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":3.7,"normalizedScore":0.8591,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-omnisciencehallucinationrate-2026-08-01","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":94.4,"normalizedScore":3.1363,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-granite-4-0-h-350m-aaifbench-2026-08-01","modelSlug":"granite-4-0-h-350m","modelName":"Granite-4.0-H-350M","providerId":"ibm","providerName":"IBM","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":17.1,"normalizedScore":2.9499,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Granite-4.0-H-350M; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aascicode-2026-08-01","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":37.6,"normalizedScore":61.8887,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-lcr-2026-08-01","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":9.7,"normalizedScore":12.8137,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aagpqadiamond-2026-08-01","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":61.5,"normalizedScore":53.6932,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aahle-2026-08-01","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.5,"normalizedScore":4.7809,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-deepseek-r1-distill-qwen-32b-aaifbench-2026-08-01","modelSlug":"deepseek-r1-distill-qwen-32b","modelName":"DeepSeek R1 Distill Qwen 32B","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":22.9,"normalizedScore":11.5044,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant DeepSeek R1 Distill Qwen 32B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-o1-pro-gpqa-2026-08-01","modelSlug":"o1-pro","modelName":"o1-pro","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"benchlm-gpqa","benchmarkName":"Graduate-Level Google-Proof Q&A","benchmarkCategory":"knowledge","benchmarkOrganisation":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jackson Petty, Richard Yuanzhe Pang, Julien Dirani, Julian Michael, Samuel R. Bowman","benchmarkVersion":"2023","score":79,"normalizedScore":76.4588,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant o1-pro; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-tau2bench-2026-08-01","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":20.5,"normalizedScore":20.6862,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aascicode-2026-08-01","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":7.4,"normalizedScore":10.9612,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-lcr-2026-08-01","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-critpt-2026-08-01","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aagpqadiamond-2026-08-01","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":42.4,"normalizedScore":26.5625,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aahle-2026-08-01","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":5.8,"normalizedScore":5.3785,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aaomniscienceindex-2026-08-01","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-82.6,"normalizedScore":3.6107,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-omniscienceaccuracy-2026-08-01","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.7,"normalizedScore":2.5773,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-omnisciencehallucinationrate-2026-08-01","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.5,"normalizedScore":6.6345,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-exaone-4-0-1-2b-aaifbench-2026-08-01","modelSlug":"exaone-4-0-1-2b","modelName":"Exaone 4.0 1.2B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":25.3,"normalizedScore":15.0442,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Exaone 4.0 1.2B; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-tau2bench-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":74.3,"normalizedScore":74.9748,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aaagenticindex-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-aaagenticindex","benchmarkName":"Artificial Analysis Agentic Index","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":8,"normalizedScore":14.0571,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-gdpvalaanormalized-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-gdpvalaanormalized","benchmarkName":"GDPval-AA normalized","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":4.9,"normalizedScore":7.1953,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-gdpvalaa-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"2026","score":598,"normalizedScore":36.1938,"unit":"elo","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aascicode-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":35.6,"normalizedScore":58.516,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aacodingindex-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-aacodingindex","benchmarkName":"Artificial Analysis Coding Index","benchmarkCategory":"coding","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":32.11,"normalizedScore":35.1661,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-lcr-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":55.7,"normalizedScore":73.5799,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-critpt-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":1.1,"normalizedScore":3.4056,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aagpqadiamond-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":78.3,"normalizedScore":77.5568,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aahle-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":13.1,"normalizedScore":19.9203,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aaomniscienceindex-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-57.9,"normalizedScore":22.9984,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-omniscienceaccuracy-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":16.5,"normalizedScore":22.8522,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-omnisciencehallucinationrate-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":89.1,"normalizedScore":9.5296,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-k-exaone-aaifbench-2026-08-01","modelSlug":"k-exaone","modelName":"K-Exaone","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":64.7,"normalizedScore":73.1563,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant K-Exaone; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-tau2bench-2026-08-01","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"tau2-bench","benchmarkName":"τ²-Bench Telecom","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"2025","score":31.9,"normalizedScore":32.1897,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aascicode-2026-08-01","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"2026","score":24.8,"normalizedScore":40.3035,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-lcr-2026-08-01","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-critpt-2026-08-01","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"critpt","benchmarkName":"CritPt","benchmarkCategory":"reasoning","benchmarkOrganisation":"CritPt authors","benchmarkVersion":"2026","score":0,"normalizedScore":0,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aagpqadiamond-2026-08-01","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"2026","score":56.1,"normalizedScore":46.0227,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aahle-2026-08-01","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"2026","score":3.8,"normalizedScore":1.3944,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aaomniscienceindex-2026-08-01","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"aa-omniscience","benchmarkName":"AA-Omniscience","benchmarkCategory":"research","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":-61.7,"normalizedScore":20.0157,"unit":"index","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"composite","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-omniscienceaccuracy-2026-08-01","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"benchlm-omniscienceaccuracy","benchmarkName":"Artificial Analysis Omniscience Accuracy","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":15.6,"normalizedScore":21.3058,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-omnisciencehallucinationrate-2026-08-01","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"benchlm-omnisciencehallucinationrate","benchmarkName":"Artificial Analysis Omniscience Hallucination Rate","benchmarkCategory":"knowledge","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"2026","score":91.5,"normalizedScore":6.6345,"unit":"percent","scoreDirection":"lower","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"benchlm-ref-solar-pro-2-aaifbench-2026-08-01","modelSlug":"solar-pro-2","modelName":"Solar Pro 2","providerId":"upstage","providerName":"Upstage","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"2026","score":33.7,"normalizedScore":27.4336,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Exact BenchLM registry variant Solar Pro 2; bulk export does not retain a complete upstream harness configuration.","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"estimated","observedAt":"2026-08-01","publishedAt":"2026-08-01","sourceId":"benchlm-public-dataset-2026-08-01","sourceTitle":"BenchLM public datasets — 2026-08-01","sourcePublisher":"BenchLM","sourceUrl":"https://benchlm.ai/data/leaderboard.json","sourceDate":"2026-08-01","sourceCheckedAt":"2026-08-01","checkedAt":"2026-08-01","verificationStatus":"source-checked"},{"resultId":"evidence-2026-08-muse-spark-1-2-terminal-bench-2-1","modelSlug":"muse-spark-1-2","modelName":"Muse Spark 1.2","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":82.9,"normalizedScore":82.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.2 (xhigh) with Muse Code","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-05","publishedAt":"2026-08-05","sourceId":"meta-muse-spark-1-2-methodology-2026-08-05","sourceTitle":"Muse Spark 1.2 Evaluation Methodology","sourcePublisher":"Meta Superintelligence Labs","sourceUrl":"https://research.meta.ai/static/muse-spark-1-2-methodology","sourceDate":"2026-08-05","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-spark-1-2-deepswe-1-1","modelSlug":"muse-spark-1-2","modelName":"Muse Spark 1.2","providerId":"meta","providerName":"Meta","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"1.1","score":59.3,"normalizedScore":59.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.2 (xhigh) with Muse Code","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-05","publishedAt":"2026-08-05","sourceId":"meta-muse-spark-1-2-methodology-2026-08-05","sourceTitle":"Muse Spark 1.2 Evaluation Methodology","sourcePublisher":"Meta Superintelligence Labs","sourceUrl":"https://research.meta.ai/static/muse-spark-1-2-methodology","sourceDate":"2026-08-05","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-spark-1-2-gdpval-aa-v2","modelSlug":"muse-spark-1-2","modelName":"Muse Spark 1.2","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":56.55,"normalizedScore":56.55,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.2 (xhigh)","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-05","publishedAt":"2026-08-05","sourceId":"aa-gdpval-muse-spark-1-2-2026-08-05","sourceTitle":"GDPval-AA v2 leaderboard — Muse Spark 1.2","sourcePublisher":"Artificial Analysis","sourceUrl":"https://artificialanalysis.ai/evaluations/gdpval-aa","sourceDate":"2026-08-05","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"independently-verified"},{"resultId":"evidence-2026-08-muse-spark-1-2-mcp-atlas","modelSlug":"muse-spark-1-2","modelName":"Muse Spark 1.2","providerId":"meta","providerName":"Meta","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"May 2026","score":90.3,"normalizedScore":90.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.2 (xhigh) in the Scale AI MCP Atlas harness","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-05","publishedAt":"2026-08-05","sourceId":"meta-muse-spark-1-2-methodology-2026-08-05","sourceTitle":"Muse Spark 1.2 Evaluation Methodology","sourcePublisher":"Meta Superintelligence Labs","sourceUrl":"https://research.meta.ai/static/muse-spark-1-2-methodology","sourceDate":"2026-08-05","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-spark-1-2-meta-internal-coding","modelSlug":"muse-spark-1-2","modelName":"Muse Spark 1.2","providerId":"meta","providerName":"Meta","benchmarkSlug":"meta-internal-coding-bench","benchmarkName":"Meta Internal Coding Bench","benchmarkCategory":"coding","benchmarkOrganisation":"Meta","benchmarkVersion":"August 2026","score":70.6,"normalizedScore":70.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Spark 1.2 (xhigh) in Meta's internal harness","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-05","publishedAt":"2026-08-05","sourceId":"meta-muse-spark-1-2-methodology-2026-08-05","sourceTitle":"Muse Spark 1.2 Evaluation Methodology","sourcePublisher":"Meta Superintelligence Labs","sourceUrl":"https://research.meta.ai/static/muse-spark-1-2-methodology","sourceDate":"2026-08-05","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-qwen-3-8-max-terminal-bench-2-1","modelSlug":"qwen-3-8-max","modelName":"Qwen3.8 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":86.6,"normalizedScore":86.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.8 Max (xhigh, default reasoning effort)","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-03","publishedAt":"2026-08-03","sourceId":"qwen-qwen38-release-2026-08-03","sourceTitle":"Qwen3.8 Max official release","sourcePublisher":"Qwen","sourceUrl":"https://qwen.ai/blog?id=qwen3.8","sourceDate":"2026-08-03","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-qwen-3-8-max-swe-bench-pro","modelSlug":"qwen-3-8-max","modelName":"Qwen3.8 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1","score":67.7,"normalizedScore":67.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.8 Max (xhigh, default reasoning effort)","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-03","publishedAt":"2026-08-03","sourceId":"qwen-qwen38-release-2026-08-03","sourceTitle":"Qwen3.8 Max official release","sourcePublisher":"Qwen","sourceUrl":"https://qwen.ai/blog?id=qwen3.8","sourceDate":"2026-08-03","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-qwen-3-8-max-deepswe-1-1","modelSlug":"qwen-3-8-max","modelName":"Qwen3.8 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"deepswe","benchmarkName":"DeepSWE","benchmarkCategory":"coding","benchmarkOrganisation":"DataCurve","benchmarkVersion":"1.1","score":56.6,"normalizedScore":56.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.8 Max (xhigh, default reasoning effort)","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-03","publishedAt":"2026-08-03","sourceId":"qwen-qwen38-release-2026-08-03","sourceTitle":"Qwen3.8 Max official release","sourcePublisher":"Qwen","sourceUrl":"https://qwen.ai/blog?id=qwen3.8","sourceDate":"2026-08-03","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-qwen-3-8-max-paperbench","modelSlug":"qwen-3-8-max","modelName":"Qwen3.8 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"paperbench","benchmarkName":"PaperBench","benchmarkCategory":"research","benchmarkOrganisation":"OpenAI","benchmarkVersion":null,"score":93,"normalizedScore":93,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.8 Max (xhigh, default reasoning effort)","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-03","publishedAt":"2026-08-03","sourceId":"qwen-qwen38-release-2026-08-03","sourceTitle":"Qwen3.8 Max official release","sourcePublisher":"Qwen","sourceUrl":"https://qwen.ai/blog?id=qwen3.8","sourceDate":"2026-08-03","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-qwen-3-8-max-jobbench","modelSlug":"qwen-3-8-max","modelName":"Qwen3.8 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"benchlm-jobbench","benchmarkName":"JobBench","benchmarkCategory":"agents","benchmarkOrganisation":"Yuetai Li, Yichen Feng, Zhangchen Xu, Zixian Ma, Kaiyuan Zheng, Fengqing Jiang, Xinghua Sun, Rulin Shao, Zichen Chen, Yue Huang, Xinyang Han, Brian Lee, Kayla Xu, Shenglai Zeng, Hang Hua, Xiangliang Zhang, Basel Alomair, Ranjay Krishna, Luke Zettlemoyer, Pang Wei Koh, Bhaskar Ramasubramanian, Luyao Niu, Xiang Yue, Radha Poovendran","benchmarkVersion":"2026","score":53.4,"normalizedScore":53.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.8 Max (xhigh, default reasoning effort)","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-03","publishedAt":"2026-08-03","sourceId":"qwen-qwen38-release-2026-08-03","sourceTitle":"Qwen3.8 Max official release","sourcePublisher":"Qwen","sourceUrl":"https://qwen.ai/blog?id=qwen3.8","sourceDate":"2026-08-03","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-qwen-3-8-max-gpqa-diamond","modelSlug":"qwen-3-8-max","modelName":"Qwen3.8 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":null,"score":92.6,"normalizedScore":92.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.8 Max (xhigh, default reasoning effort)","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-03","publishedAt":"2026-08-03","sourceId":"qwen-qwen38-release-2026-08-03","sourceTitle":"Qwen3.8 Max official release","sourcePublisher":"Qwen","sourceUrl":"https://qwen.ai/blog?id=qwen3.8","sourceDate":"2026-08-03","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-qwen-3-8-max-hle-no-tools","modelSlug":"qwen-3-8-max","modelName":"Qwen3.8 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":null,"score":43.6,"normalizedScore":43.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.8 Max (xhigh, default reasoning effort)","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-03","publishedAt":"2026-08-03","sourceId":"qwen-qwen38-release-2026-08-03","sourceTitle":"Qwen3.8 Max official release","sourcePublisher":"Qwen","sourceUrl":"https://qwen.ai/blog?id=qwen3.8","sourceDate":"2026-08-03","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-qwen-3-8-max-ifbench","modelSlug":"qwen-3-8-max","modelName":"Qwen3.8 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":null,"score":82.8,"normalizedScore":82.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.8 Max (xhigh, default reasoning effort)","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-03","publishedAt":"2026-08-03","sourceId":"qwen-qwen38-release-2026-08-03","sourceTitle":"Qwen3.8 Max official release","sourcePublisher":"Qwen","sourceUrl":"https://qwen.ai/blog?id=qwen3.8","sourceDate":"2026-08-03","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-qwen-3-8-max-osworld-verified","modelSlug":"qwen-3-8-max","modelName":"Qwen3.8 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified","score":86.1,"normalizedScore":86.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.8 Max (xhigh, default reasoning effort)","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-03","publishedAt":"2026-08-03","sourceId":"qwen-qwen38-release-2026-08-03","sourceTitle":"Qwen3.8 Max official release","sourcePublisher":"Qwen","sourceUrl":"https://qwen.ai/blog?id=qwen3.8","sourceDate":"2026-08-03","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-qwen-3-8-max-mmmu-pro","modelSlug":"qwen-3-8-max","modelName":"Qwen3.8 Max","providerId":"alibaba","providerName":"Alibaba Cloud","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":null,"score":82.3,"normalizedScore":82.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Qwen3.8 Max (xhigh, default reasoning effort)","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-03","publishedAt":"2026-08-03","sourceId":"qwen-qwen38-release-2026-08-03","sourceTitle":"Qwen3.8 Max official release","sourcePublisher":"Qwen","sourceUrl":"https://qwen.ai/blog?id=qwen3.8","sourceDate":"2026-08-03","sourceCheckedAt":"2026-08-05","checkedAt":"2026-08-05","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-mcp-atlas-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"mcp-atlas","benchmarkName":"MCP Atlas","benchmarkCategory":"agents","benchmarkOrganisation":"Google","benchmarkVersion":"Public / 500 tasks","score":75.5,"normalizedScore":75.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-tau3-banking-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"tau3-banking","benchmarkName":"τ³-Banking","benchmarkCategory":"agents","benchmarkOrganisation":"Sierra / Artificial Analysis","benchmarkVersion":"3 / Banking","score":23.5,"normalizedScore":23.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-osworld-verified-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"osworld-verified","benchmarkName":"OSWorld-Verified","benchmarkCategory":"agents","benchmarkOrganisation":"OSWorld","benchmarkVersion":"Verified / 361 tasks","score":65.9,"normalizedScore":65.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-swe-bench-pro-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-pro","benchmarkName":"SWE-Bench Pro","benchmarkCategory":"coding","benchmarkOrganisation":"Scale AI","benchmarkVersion":"v1 / original 731-task set","score":51.2,"normalizedScore":51.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-terminal-bench-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"terminal-bench","benchmarkName":"Terminal-Bench","benchmarkCategory":"agents","benchmarkOrganisation":"Terminal-Bench team","benchmarkVersion":"2.1","score":51.7,"normalizedScore":51.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-charxiv-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"charxiv","benchmarkName":"CharXiv","benchmarkCategory":"multimodal","benchmarkOrganisation":"CharXiv","benchmarkVersion":"Reasoning / validation set","score":78.8,"normalizedScore":78.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-mmmu-pro-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"1,730-question set","score":74,"normalizedScore":74,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-aime-2026-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"aime-2026","benchmarkName":"AIME 2026","benchmarkCategory":"mathematics","benchmarkOrganisation":"MAA / public contest evals","benchmarkVersion":"2026 / 30 questions","score":94.7,"normalizedScore":94.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-gpqa-diamond-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"gpqa-diamond","benchmarkName":"GPQA Diamond","benchmarkCategory":"reasoning","benchmarkOrganisation":"GPQA authors","benchmarkVersion":"198-question Diamond set","score":83.5,"normalizedScore":83.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-humanitys-last-exam-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"humanitys-last-exam","benchmarkName":"Humanity's Last Exam","benchmarkCategory":"reasoning","benchmarkOrganisation":"Center for AI Safety and Scale AI","benchmarkVersion":"text-only / no tools / 2,158 questions","score":22,"normalizedScore":22,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-aa-lcr-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"aa-lcr","benchmarkName":"AA Long Context Reasoning","benchmarkCategory":"reasoning","benchmarkOrganisation":"Artificial Analysis","benchmarkVersion":"100-question set","score":80,"normalizedScore":80,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-gdpval-aa-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"gdpval-aa","benchmarkName":"GDPval-AA v2","benchmarkCategory":"agents","benchmarkOrganisation":"Artificial Analysis / OpenAI","benchmarkVersion":"v2","score":953,"normalizedScore":22.65,"unit":"Elo","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-benchlm-deepsearchqa-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-deepsearchqa","benchmarkName":"DeepSearchQA","benchmarkCategory":"agents","benchmarkOrganisation":"Meta AI","benchmarkVersion":"900 questions","score":74.6,"normalizedScore":74.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-benchlm-skillsbench-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-skillsbench","benchmarkName":"Vals SkillsBench","benchmarkCategory":"knowledge","benchmarkOrganisation":"Vals AI","benchmarkVersion":"86 tasks / with skills","score":44.3,"normalizedScore":44.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-swe-bench-verified-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"swe-bench-verified","benchmarkName":"SWE-bench Verified","benchmarkCategory":"coding","benchmarkOrganisation":"SWE-bench authors","benchmarkVersion":"Verified / 500 tasks","score":76,"normalizedScore":76,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-scicode-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"scicode","benchmarkName":"SciCode","benchmarkCategory":"coding","benchmarkOrganisation":"SciCode authors","benchmarkVersion":"288 test subproblems","score":43.6,"normalizedScore":43.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-screenspot-pro-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"screenspot-pro","benchmarkName":"ScreenSpot-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"ScreenSpot","benchmarkVersion":"Pro / 1,581 examples","score":75.4,"normalizedScore":75.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-benchlm-omnidocbench15-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"benchlm-omnidocbench15","benchmarkName":"OmniDocBench 1.5","benchmarkCategory":"multimodal","benchmarkOrganisation":"OpenAI","benchmarkVersion":"v1.5 / 1,355 prompts","score":75.8,"normalizedScore":75.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"evidence-2026-08-muse-glimmer-30b-ifbench-high","modelSlug":"muse-glimmer-30b","modelName":"Muse Glimmer 30B","providerId":"meta","providerName":"Meta","benchmarkSlug":"ifbench","benchmarkName":"IFBench","benchmarkCategory":"instruction-following","benchmarkOrganisation":"IFBench","benchmarkVersion":"294 tasks","score":77,"normalizedScore":77,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Muse Glimmer-30B; high reasoning; temperature=1.0; top_p=0.95; top_k=64","methodologyVersion":"1.8.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-10","publishedAt":"2026-08-10","sourceId":"meta-muse-glimmer-30b-methodology-2026-08-10","sourceTitle":"Muse Glimmer Evaluation Methodology","sourcePublisher":"Meta Superintelligence Lab","sourceUrl":"https://research.meta.ai/static/muse-glimmer-methodology","sourceDate":"2026-08-10","sourceCheckedAt":"2026-08-10","checkedAt":"2026-08-10","verificationStatus":"provider-reported"},{"resultId":"livecodebench-lcb:deepseek-v3:1722470400000-1746057600000","modelSlug":"deepseek-v3","modelName":"DeepSeek V3","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"standard","score":27.2,"normalizedScore":27.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek-V3","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2025-05-01","publishedAt":"2025-05-01","sourceId":"refresh-livecodebench","sourceTitle":"LiveCodeBench permanent refresh source","sourcePublisher":"LiveCodeBench","sourceUrl":"https://livecodebench.github.io/leaderboard.html","sourceDate":"2025-05-01","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"livecodebench-lcb:gpt-4-turbo-2024-04-09:1722470400000-1746057600000","modelSlug":"gpt-4-turbo","modelName":"GPT-4 Turbo","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"standard","score":28.7,"normalizedScore":28.7,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-4-Turbo-2024-04-09","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2025-05-01","publishedAt":"2025-05-01","sourceId":"refresh-livecodebench","sourceTitle":"LiveCodeBench permanent refresh source","sourcePublisher":"LiveCodeBench","sourceUrl":"https://livecodebench.github.io/leaderboard.html","sourceDate":"2025-05-01","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"livecodebench-lcb:gpt-4o-2024-08-06:1722470400000-1746057600000","modelSlug":"gpt-4o-2024-08-06","modelName":"GPT-4o (Aug '24)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"standard","score":29.5,"normalizedScore":29.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-4O-2024-08-06","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2025-05-01","publishedAt":"2025-05-01","sourceId":"refresh-livecodebench","sourceTitle":"LiveCodeBench permanent refresh source","sourcePublisher":"LiveCodeBench","sourceUrl":"https://livecodebench.github.io/leaderboard.html","sourceDate":"2025-05-01","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"livecodebench-lcb:gpt-4o-mini-2024-07-18:1722470400000-1746057600000","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"standard","score":27.5,"normalizedScore":27.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-4O-mini-2024-07-18","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2025-05-01","publishedAt":"2025-05-01","sourceId":"refresh-livecodebench","sourceTitle":"LiveCodeBench permanent refresh source","sourcePublisher":"LiveCodeBench","sourceUrl":"https://livecodebench.github.io/leaderboard.html","sourceDate":"2025-05-01","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"livecodebench-lcb:claude-3-5-sonnet-20241022:1722470400000-1746057600000","modelSlug":"claude-35-sonnet","modelName":"Claude 3.5 Sonnet (Oct '24)","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"standard","score":36.4,"normalizedScore":36.4,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude-3.5-Sonnet-20241022","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2025-05-01","publishedAt":"2025-05-01","sourceId":"refresh-livecodebench","sourceTitle":"LiveCodeBench permanent refresh source","sourcePublisher":"LiveCodeBench","sourceUrl":"https://livecodebench.github.io/leaderboard.html","sourceDate":"2025-05-01","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"livecodebench-lcb:claude-3-haiku:1722470400000-1746057600000","modelSlug":"claude-3-haiku","modelName":"Claude 3 Haiku","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"standard","score":20.2,"normalizedScore":20.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude-3-Haiku","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2025-05-01","publishedAt":"2025-05-01","sourceId":"refresh-livecodebench","sourceTitle":"LiveCodeBench permanent refresh source","sourcePublisher":"LiveCodeBench","sourceUrl":"https://livecodebench.github.io/leaderboard.html","sourceDate":"2025-05-01","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"livecodebench-lcb:claude-opus-4:1722470400000-1746057600000","modelSlug":"claude-opus-4","modelName":"Claude Opus 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"standard","score":46.9,"normalizedScore":46.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude-Opus-4","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2025-05-01","publishedAt":"2025-05-01","sourceId":"refresh-livecodebench","sourceTitle":"LiveCodeBench permanent refresh source","sourcePublisher":"LiveCodeBench","sourceUrl":"https://livecodebench.github.io/leaderboard.html","sourceDate":"2025-05-01","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"livecodebench-lcb:claude-sonnet-4:1722470400000-1746057600000","modelSlug":"claude-sonnet-4","modelName":"Claude Sonnet 4","providerId":"anthropic","providerName":"Anthropic","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"standard","score":47.1,"normalizedScore":47.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Claude-Sonnet-4","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2025-05-01","publishedAt":"2025-05-01","sourceId":"refresh-livecodebench","sourceTitle":"LiveCodeBench permanent refresh source","sourcePublisher":"LiveCodeBench","sourceUrl":"https://livecodebench.github.io/leaderboard.html","sourceDate":"2025-05-01","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"livecodebench-lcb:gemini-2-5-pro-05-06:1722470400000-1746057600000","modelSlug":"gemini-2-5-pro-05-06","modelName":"Gemini 2.5 Pro Preview (May' 25)","providerId":"google","providerName":"Google","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"standard","score":71.8,"normalizedScore":71.8,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini-2.5-Pro-05-06","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2025-05-01","publishedAt":"2025-05-01","sourceId":"refresh-livecodebench","sourceTitle":"LiveCodeBench permanent refresh source","sourcePublisher":"LiveCodeBench","sourceUrl":"https://livecodebench.github.io/leaderboard.html","sourceDate":"2025-05-01","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"livecodebench-lcb:deepseek-r1-0528:1722470400000-1746057600000","modelSlug":"deepseek-r1-aa-2","modelName":"DeepSeek R1 0528 (May '25)","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"standard","score":73.1,"normalizedScore":73.1,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek-R1-0528","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2025-05-01","publishedAt":"2025-05-01","sourceId":"refresh-livecodebench","sourceTitle":"LiveCodeBench permanent refresh source","sourcePublisher":"LiveCodeBench","sourceUrl":"https://livecodebench.github.io/leaderboard.html","sourceDate":"2025-05-01","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"livecodebench-lcb:exaone-4-0-32b:1722470400000-1746057600000","modelSlug":"exaone-4-0-32b","modelName":"Exaone 4.0 32B","providerId":"lg-ai-research","providerName":"LG AI Research","benchmarkSlug":"livecodebench","benchmarkName":"LiveCodeBench","benchmarkCategory":"coding","benchmarkOrganisation":"LiveCodeBench team","benchmarkVersion":"standard","score":70,"normalizedScore":70,"unit":"percent","scoreDirection":"higher","modelConfiguration":"EXAONE-4.0-32B","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2025-05-01","publishedAt":"2025-05-01","sourceId":"refresh-livecodebench","sourceTitle":"LiveCodeBench permanent refresh source","sourcePublisher":"LiveCodeBench","sourceUrl":"https://livecodebench.github.io/leaderboard.html","sourceDate":"2025-05-01","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"mmmu-pro-owner-gpt-4o-0513","modelSlug":"gpt-4o-2024-05-13","modelName":"GPT-4o (May '24)","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":51.9,"normalizedScore":51.9,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-4o (0513)","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2024-05-13","publishedAt":"2024-05-13","sourceId":"refresh-mmmu","sourceTitle":"MMMU permanent refresh source","sourcePublisher":"MMMU","sourceUrl":"https://mmmu-benchmark.github.io/","sourceDate":"2024-05-13","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"mmmu-pro-owner-gemini-1-5-pro-0523","modelSlug":"gemini-1-5-pro-may-2024","modelName":"Gemini 1.5 Pro (May '24)","providerId":"google","providerName":"Google","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":43.5,"normalizedScore":43.5,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Gemini 1.5 Pro (0523)","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2024-05-23","publishedAt":"2024-05-23","sourceId":"refresh-mmmu","sourceTitle":"MMMU permanent refresh source","sourcePublisher":"MMMU","sourceUrl":"https://mmmu-benchmark.github.io/","sourceDate":"2024-05-23","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"mmmu-pro-owner-gpt-4o-mini","modelSlug":"gpt-4o-mini","modelName":"GPT-4o mini","providerId":"openai","providerName":"OpenAI","benchmarkSlug":"mmmu-pro","benchmarkName":"MMMU-Pro","benchmarkCategory":"multimodal","benchmarkOrganisation":"MMMU-Pro authors","benchmarkVersion":"2024","score":37.6,"normalizedScore":37.6,"unit":"percent","scoreDirection":"higher","modelConfiguration":"GPT-4o mini","methodologyVersion":"2.2.0","inclusionStatus":"ranking-eligible","evidenceState":"direct","observedAt":"2024-07-18","publishedAt":"2024-07-18","sourceId":"refresh-mmmu","sourceTitle":"MMMU permanent refresh source","sourcePublisher":"MMMU","sourceUrl":"https://mmmu-benchmark.github.io/","sourceDate":"2024-07-18","sourceCheckedAt":"2026-08-07","checkedAt":"2026-08-18","verificationStatus":"official-leaderboard"},{"resultId":"deepseek-v4-flash-vision-exp-agents-last-exam-2026-08-21","modelSlug":"deepseek-v4-flash-vision-exp","modelName":"DeepSeek V4 Flash Vision Exp","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-alebench","benchmarkName":"Agents Last Exam","benchmarkCategory":"knowledge","benchmarkOrganisation":"UC Berkeley RDI","benchmarkVersion":"2026","score":27.3,"normalizedScore":27.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Provider chart does not specify an exact evaluation system","methodologyVersion":"2.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-21","publishedAt":"2026-08-21","sourceId":"deepseek-v4-flash-vision-exp-release","sourceTitle":"DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","sourcePublisher":"DeepSeek","sourceUrl":"https://api-docs.deepseek.com/zh-cn/updates/","sourceDate":"2026-08-21","sourceCheckedAt":"2026-08-27","checkedAt":"2026-08-24","verificationStatus":"provider-reported"},{"resultId":"deepseek-v4-flash-vision-exp-chartography-2026-08-21","modelSlug":"deepseek-v4-flash-vision-exp","modelName":"DeepSeek V4 Flash Vision Exp","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-chartography","benchmarkName":"Chartography without tools","benchmarkCategory":"multimodal","benchmarkOrganisation":"Surge AI and Anthropic","benchmarkVersion":"2026","score":64.3,"normalizedScore":64.3,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Provider chart does not specify tool use or evaluation settings","methodologyVersion":"2.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-21","publishedAt":"2026-08-21","sourceId":"deepseek-v4-flash-vision-exp-release","sourceTitle":"DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","sourcePublisher":"DeepSeek","sourceUrl":"https://api-docs.deepseek.com/zh-cn/updates/","sourceDate":"2026-08-21","sourceCheckedAt":"2026-08-27","checkedAt":"2026-08-24","verificationStatus":"provider-reported"},{"resultId":"deepseek-v4-flash-vision-exp-zerobench-pass-at-5-2026-08-21","modelSlug":"deepseek-v4-flash-vision-exp","modelName":"DeepSeek V4 Flash Vision Exp","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"zerobench","benchmarkName":"ZeroBench","benchmarkCategory":"multimodal","benchmarkOrganisation":"ZeroBench","benchmarkVersion":"August 2026","score":35,"normalizedScore":35,"unit":"percent","scoreDirection":"higher","modelConfiguration":"Provider chart specifies Pass@5 but not sampling or tool settings","methodologyVersion":"2.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-08-21","publishedAt":"2026-08-21","sourceId":"deepseek-v4-flash-vision-exp-release","sourceTitle":"DeepSeek-V4-Flash-Vision-Exp release and provider evaluation","sourcePublisher":"DeepSeek","sourceUrl":"https://api-docs.deepseek.com/zh-cn/updates/","sourceDate":"2026-08-21","sourceCheckedAt":"2026-08-27","checkedAt":"2026-08-24","verificationStatus":"provider-reported"},{"resultId":"deepseek-v4-flash-0731-agents-last-exam-2026-07-31","modelSlug":"deepseek-v4-flash-0731","modelName":"DeepSeek V4 Flash 0731","providerId":"deepseek","providerName":"DeepSeek","benchmarkSlug":"benchlm-alebench","benchmarkName":"Agents Last Exam","benchmarkCategory":"knowledge","benchmarkOrganisation":"UC Berkeley RDI","benchmarkVersion":"2026","score":25.2,"normalizedScore":25.2,"unit":"percent","scoreDirection":"higher","modelConfiguration":"DeepSeek-V4-Flash-0731 official model card; exact ALE configuration not specified","methodologyVersion":"2.3.0","inclusionStatus":"reference-only","evidenceState":"direct","observedAt":"2026-07-31","publishedAt":"2026-07-31","sourceId":"deepseek-v4-flash-0731-model-card","sourceTitle":"DeepSeek-V4-Flash-0731 official model card and provider evaluation","sourcePublisher":"DeepSeek","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","sourceDate":"2026-07-31","sourceCheckedAt":"2026-08-24","checkedAt":"2026-08-24","verificationStatus":"provider-reported"}]}
